diff --git a/.github/workflows/wiki-sync.yml b/.github/workflows/wiki-sync.yml
new file mode 100644
index 00000000000..7381ed56483
--- /dev/null
+++ b/.github/workflows/wiki-sync.yml
@@ -0,0 +1,69 @@
+name: Wiki Sync
+
+# Keeps the GitHub wiki in sync with docs/ on every release that lands on main.
+# The wiki has no native generator and historically drifts (it sat at "212+ providers /
+# 14 strategies / 37 MCP tools" while code was at 226 / 15 / 87, and new docs like
+# SUPPLY_CHAIN never appeared). This runs scripts/docs/sync-wiki.mjs, which:
+# - ADDS any docs/ page missing from the wiki (curated; internal reports excluded),
+# - syncs the four cover-page counts on Home.md.
+# It does NOT overwrite existing wiki pages by default: several docs sources still carry
+# stale counts (e.g. ARCHITECTURE.md says "177 providers" while the wiki cover is 226),
+# so blind overwrite would regress the wiki. Full content parity (--update-existing) is
+# gated on regenerating those sources — see docs/ops/DOCUMENTATION_AUDIT_REPORT.md.
+
+on:
+ push:
+ branches: [main]
+ paths:
+ - "docs/**"
+ - "README.md"
+ - "AGENTS.md"
+ - "src/shared/constants/routingStrategies.ts"
+ - "config/i18n.json"
+ - "open-sse/mcp-server/server.ts"
+ - "scripts/docs/sync-wiki.mjs"
+ workflow_dispatch:
+
+permissions:
+ contents: write
+
+concurrency:
+ group: wiki-sync
+ cancel-in-progress: false
+
+jobs:
+ sync-wiki:
+ name: Sync wiki with docs
+ runs-on: ubuntu-latest
+ steps:
+ - name: Checkout repo
+ uses: actions/checkout@v4
+
+ - name: Setup Node
+ uses: actions/setup-node@v4
+ with:
+ node-version: "24"
+
+ - name: Clone wiki
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ REPO: ${{ github.repository }}
+ run: |
+ git clone "https://x-access-token:${GH_TOKEN}@github.com/${REPO}.wiki.git" wiki
+
+ - name: Sync wiki (add missing pages + cover counts)
+ run: node scripts/docs/sync-wiki.mjs --wiki-dir wiki
+
+ - name: Commit & push if changed
+ run: |
+ cd wiki
+ if [ -n "$(git status --porcelain)" ]; then
+ git config user.name "github-actions[bot]"
+ git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
+ git add -A
+ git commit -m "docs(wiki): auto-sync pages + cover counts with docs"
+ git push
+ echo "Wiki updated."
+ else
+ echo "Wiki already in sync — nothing to push."
+ fi
diff --git a/.gitignore b/.gitignore
index cbe3af7f64d..1a94d63b2b7 100644
--- a/.gitignore
+++ b/.gitignore
@@ -203,8 +203,11 @@ pr_reviews*.json
# internal setup prompts with personal credentials — never commit
CODEX-SETUP-PROMPT.md
-# Quality ratchet — métricas efêmeras (baseline é commitado, métricas não)
-quality-metrics.json
+# Quality ratchet — métricas efêmeras (baseline commitado em config/quality/; métricas não)
+config/quality/quality-metrics.json
+
+# Runtime logs (diretório local, nunca versionado)
+/logs/
-home-diegosouzapw-dev-automações-bots-yt-downloader-20260504 .txt
-home-diegosouzapw-dev-automações-bots-yt-downloader-20260410 .txt
docs/prompts/AGENT-OWNERSHIP-PROTOCOL.omniroute.md
@@ -212,3 +215,4 @@ docs/prompts/AGENT-OWNERSHIP-PROTOCOL.md
docs/prompts/AGENT-OWNERSHIP-PROTOCOL.omniroute-mim.md
docs/prompts/AGENT-OWNERSHIP-PROTOCOL.omniroute-mid.md
omniroute.md
+quality-metrics.json
diff --git a/CHANGELOG.md b/CHANGELOG.md
index bcdcd4e4b5a..b6aa2eb465b 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,6 +4,15 @@
---
+## [3.8.26] — TBD
+
+### 🧹 Internal / Quality / Docs
+
+- **fix(ci): grant `contents: write` to the npm publish job for SBOM attach** — the v3.8.25 TokenPermissions hardening set the npm-publish `publish` job to `contents: read`, but its "Attach SBOM to GitHub Release" step (`gh release upload`) needs `contents: write` and failed with HTTP 403 on the v3.8.25 release (npm / GitHub Packages / opencode-plugin / Docker / Electron all published fine; only the SBOM attach broke — the v3.8.25 SBOM was attached manually). ([#3874](https://github.com/diegosouzapw/OmniRoute/pull/3874) — thanks @diegosouzapw)
+- **docs: refresh the provider count to 226 + regenerate `PROVIDER_REFERENCE.md`** — the README advertised a stale `177 providers`; the canonical generator (`scripts/docs/gen-provider-reference.ts`) now reports **226 unique provider IDs**, so the README badges/anchors and the generated provider reference were brought in sync. Also adds a documentation audit/sync report. (thanks @diegosouzapw)
+
+---
+
## [3.8.25] — 2026-06-14
### ✨ New Features
diff --git a/README.md b/README.md
index 950f1b27302..b2783bea126 100644
--- a/README.md
+++ b/README.md
@@ -6,7 +6,7 @@
# 🚀 OmniRoute — The Free AI Gateway
-### Never stop coding. Connect every AI tool to **177 providers** — **50+ free** — through one endpoint.
+### Never stop coding. Connect every AI tool to **226 providers** — **50+ free** — through one endpoint.
**Plug Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini. Auto-fallback.**
@@ -19,8 +19,8 @@
-[](#-177-ai-providers--50-free)
-[](#-177-ai-providers--50-free)
+[](#-226-ai-providers--50-free)
+[](#-226-ai-providers--50-free)
[](docs/reference/FREE_TIERS.md)
[](#%EF%B8%8F-save-1595-tokens--automatically)
[](#-combos--the-flagship)
@@ -59,7 +59,7 @@
-[**🚀 Quick Start**](#-quick-start) • [**🎯 Combos**](#-combos--the-flagship) • [**🌐 Providers**](#-177-ai-providers--50-free) • [**🔌 CLI & MCP**](#-full-cli--a2a--mcp) • [**🗜️ Compression**](#%EF%B8%8F-save-1595-tokens--automatically) • [**🌍 Website**](https://omniroute.online)
+[**🚀 Quick Start**](#-quick-start) • [**🎯 Combos**](#-combos--the-flagship) • [**🌐 Providers**](#-226-ai-providers--50-free) • [**🔌 CLI & MCP**](#-full-cli--a2a--mcp) • [**🗜️ Compression**](#%EF%B8%8F-save-1595-tokens--automatically) • [**🌍 Website**](https://omniroute.online)
[💥 The Promise](#-the-promise) • [🤔 Why](#-why-omniroute) • [🏆 What Sets Apart](#-what-sets-omniroute-apart) • [🤖 Compatible CLIs](#-compatible-clis--coding-agents) • [🖥️ Where It Runs](#%EF%B8%8F-where-omniroute-runs--anywhere) • [🔒 Private](#-private--local-first) • [🎬 In Action](#-omniroute-in-action) • [📚 Explore More](#-explore-more) • [📧 Support](#-support--community)
@@ -136,18 +136,18 @@
-> One endpoint. **177 providers.** Never stop building — and let OmniRoute pick the cheapest one that works.
+> One endpoint. **226 providers.** Never stop building — and let OmniRoute pick the cheapest one that works.
- 🚫 Never hit limits Auto-fallback across 177 providers in milliseconds. Quota out? Next provider takes over — zero downtime.
+ 🚫 Never hit limits Auto-fallback across 226 providers in milliseconds. Quota out? Next provider takes over — zero downtime.
💸 Save up to 95% tokens RTK + Caveman stacked compression cuts 15–95% of eligible tokens (~89% avg on tool-heavy sessions).
🆓 $0 to start 50+ providers with a free tier, 11 free forever (Kiro, Qoder, Pollinations, LongCat…). No card needed.
🔌 Every tool works 16+ coding agents — Claude Code, Codex, Cursor, Cline, Copilot, Antigravity — through one config.
🧩 One endpoint OpenAI ↔ Claude ↔ Gemini ↔ Responses API translation. Point any tool at /v1 and it just works.
- 🛡️ Production-grade Circuit breakers, TLS stealth, MCP (87 tools), A2A, memory, guardrails, evals. 4,690+ tests.
+ 🛡️ Production-grade Circuit breakers, TLS stealth, MCP (87 tools), A2A, memory, guardrails, evals. 14,965 tests.
@@ -263,7 +263,7 @@ Result: 4 layers of fallback = zero downtime
| Feature | OmniRoute | Other routers |
| -------------------------------------- | ----------------------------------------------------------- | ------------- |
-| 🌐 Providers | **177** | 20–100 |
+| 🌐 Providers | **226** | 20–100 |
| 🆓 Free providers | **50+ (11 free forever)** | 1–5 |
| 🔀 Routing strategies | **15** (priority, weighted, cost-optimized, context-relay…) | 1–3 |
| 🗜️ Token compression | **RTK + Caveman stacked (15–95%)** | None / 20–40% |
@@ -274,7 +274,7 @@ Result: 4 layers of fallback = zero downtime
| ☁️ Cloud agents | **Codex, Devin, Jules** | None |
| 🥷 TLS fingerprint stealth | **JA3/JA4 via wreq-js** | None |
| 🖥️ Multi-platform | **Web · Desktop · Termux · PWA** | Web only |
-| 🌍 i18n | **40+ locales** | 0–4 |
+| 🌍 i18n | **42 locales** | 0–4 |
📊 Detailed comparison vs LiteLLM, OpenRouter & Portkey → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md)
@@ -319,11 +319,11 @@ Result: 4 layers of fallback = zero downtime
-# 🌐 177 AI Providers — 50+ Free
+# 🌐 226 AI Providers — 50+ Free
-> The most complete catalog of any open-source router: **177 providers**, **50+ with a free tier**, **11 free forever**.
+> The most complete catalog of any open-source router: **226 providers**, **50+ with a free tier**, **11 free forever**.
@@ -439,7 +439,25 @@ claude mcp add-server omniroute --type http --url http://localhost:20128/api/mcp
-> **Why use many token when few token do trick?** Every request passes through OmniRoute's compression pipeline **transparently** — no client changes. It stacks ideas from [RTK](https://github.com/rtk-ai/rtk), [Caveman](https://github.com/JuliusBrussee/caveman) (⭐ 51K+), and [Troglodita](https://github.com/leninejunior/troglodita) (PT-BR).
+> **Why use many token when few token do trick?** Every request passes through OmniRoute's compression pipeline **transparently** — no client changes. It's now a **stack of 9 composable engines** that run in order and mix & match per routing combo — building on ideas from [RTK](https://github.com/rtk-ai/rtk), [Caveman](https://github.com/JuliusBrussee/caveman) (⭐ 51K+), [LLMLingua-2](https://github.com/microsoft/LLMLingua), and [Troglodita](https://github.com/leninejunior/troglodita) (PT-BR).
+
+### 🧱 The 9-engine stack
+
+Engines run in pipeline order; each is independently toggleable and configurable per combo:
+
+| # | Engine | What it does |
+| --- | ----------------- | ------------------------------------------------------------------------ |
+| 1 | **Session-Dedup** | Drops content repeated across turns (content-addressed, cross-turn) |
+| 2 | **CCR** | Archives large blocks behind retrieve markers, fetched on demand |
+| 3 | **RTK** | Smart tool-result filtering, dedup & truncation (command-aware) |
+| 4 | **Headroom** | Lossless tabular compaction of homogeneous JSON arrays (~30%+) |
+| 5 | **Caveman** | Rule-based prose compression (~65–75% on output) |
+| 6 | **LLMLingua-2** | ML semantic pruning via MobileBERT ONNX — code-safe, async |
+| 7 | **Lite** | Whitespace + image-URL trimming (latency-light baseline) |
+| 8 | **Aggressive** | Summarization + progressive aging of old turns |
+| 9 | **Ultra** | Heuristic token pruning with an optional small-model (SLM) tier |
+
+Code blocks, URLs and structured data are **always preserved** byte-perfect. **One-click presets** combine the engines:
| Mode | Savings | Best for |
| ------------------------------ | ---------- | --------------------------- |
@@ -471,7 +489,7 @@ claude mcp add-server omniroute --type http --url http://localhost:20128/api/mcp
### 📖 How it works — pipeline, architecture & savings math
```
-Client (10,000 tok) ──▶ OmniRoute Compression (7 options) ──▶ Provider (~1,080 tok, up to 95% saved)
+Client (10,000 tok) ──▶ OmniRoute Compression (9 engines) ──▶ Provider (~1,080 tok, up to 95% saved)
```
Default stacked combo runs `RTK → Caveman`. When both act on the same tool/context payload, savings compound:
@@ -733,7 +751,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo
**Will I be charged by OmniRoute?** No — it's free, open-source software on your machine. You only pay paid providers directly. OmniRoute has no billing system.
**Are FREE providers really unlimited?** Yes — Kiro, Qoder, Pollinations, LongCat, Cloudflare. No catch.
**Will compression hurt quality?** No — it only compresses the **input**; code, URLs, JSON are always protected.
-**Does it work where AI is blocked?** Yes — 3-level proxy + 1proxy marketplace reach all 177 providers.
+**Does it work where AI is blocked?** Yes — 3-level proxy + 1proxy marketplace reach all 226 providers.
📖 [User Guide](docs/guides/USER_GUIDE.md) · [API Reference](docs/reference/API_REFERENCE.md) · [Environment Config](docs/reference/ENVIRONMENT.md)
@@ -803,7 +821,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo
- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
- **Streaming**: Server-Sent Events (SSE) + WebSocket bridge (`/v1/ws`)
- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
-- **Testing**: Node.js test runner + Vitest (**4,690+ test cases** across 517 files — unit, integration, E2E, security, ecosystem)
+- **Testing**: Node.js test runner + Vitest (**14,965 test cases** across 517 files — unit, integration, E2E, security, ecosystem)
- **Platforms**: Desktop (Electron), Android (Termux), PWA (any browser)
- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
- **Website**: [omniroute.online](https://omniroute.online)
@@ -878,7 +896,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo
| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
| [i18n Guide](docs/guides/I18N.md) | 40+ language support, translation workflow, RTL |
| [Release Checklist](docs/ops/RELEASE_CHECKLIST.md) | Pre-release validation steps |
-| [Coverage Plan](docs/ops/COVERAGE_PLAN.md) | Test coverage strategy and 4,690+ test suite |
+| [Coverage Plan](docs/ops/COVERAGE_PLAN.md) | Test coverage strategy and 14,965 test suite |
diff --git a/.license-allowlist.json b/config/quality/.license-allowlist.json
similarity index 100%
rename from .license-allowlist.json
rename to config/quality/.license-allowlist.json
diff --git a/complexity-baseline.json b/config/quality/complexity-baseline.json
similarity index 100%
rename from complexity-baseline.json
rename to config/quality/complexity-baseline.json
diff --git a/dependency-allowlist.json b/config/quality/dependency-allowlist.json
similarity index 100%
rename from dependency-allowlist.json
rename to config/quality/dependency-allowlist.json
diff --git a/duplication-baseline.json b/config/quality/duplication-baseline.json
similarity index 100%
rename from duplication-baseline.json
rename to config/quality/duplication-baseline.json
diff --git a/file-size-baseline.json b/config/quality/file-size-baseline.json
similarity index 90%
rename from file-size-baseline.json
rename to config/quality/file-size-baseline.json
index 7dbcc932a70..142b9b8f141 100644
--- a/file-size-baseline.json
+++ b/config/quality/file-size-baseline.json
@@ -2,9 +2,13 @@
"_comment": "Catraca de tamanho (check-file-size.mjs). frozen so pode encolher; arquivos novos <= cap. --update ratcheta.",
"_rebaseline_v3.8.25": "Drift consciente do ciclo v3.8.24->v3.8.25 (features #3799-#3806: free-provider-rankings, plugins menu, proxy IP-family selector). 3 arquivos cresceram por feature legitima, nao por regressao de qualidade: ProxyRegistryManager.tsx 1072->1089, sidebarVisibility.ts 990->1006, schemas.ts 2519->2522. Encolher fica como debt para um refactor dedicado.",
"_rebaseline_2026_06_15_3860_compression_ui": "PR #3860 own growth: sidebarVisibility.ts 1006->1100 (+94 = Compression Hub menu entries: Hub + per-engine Lite/Aggressive/Ultra pages + combos editor) and chatCore.ts 5812->5815 (+3 = compression UI config wiring). Cohesive feature growth, not a quality regression.",
+ "_rebaseline_2026_06_15_3885_glm_5_2": "PR #3885 own growth: pricing.ts 1508->1529 (+21 = GLM-5.2 pricing rows for glm-5.2 + effort aliases glm-5.2-high/-max, same $1.2/$5 schedule as glm-5.1; pure data). Also adds glm-5.2 specs to glmProvider.ts/modelSpecs.ts (modelSpecs.ts stays under cap). Cohesive model registration; not extractable.",
+ "_rebaseline_2026_06_15_3870_alias_lookup": "PR #3870 own growth: providerRegistry.ts 4703->4708 (+5 = generateModels() also stores each provider's models under its raw id, not only its alias, so getProviderModels(rawId) works when alias != id e.g. github->gh; preserves the existing first-wins guard). Cohesive registry fix; not extractable.",
+ "_rebaseline_2026_06_15_3846_sticky_combo_rr": "PR #3846 own growth: combo.ts 5204->5277 (+73 = combo-level sticky round-robin reusing the existing global stickyRoundRobinLimit knob #3847 added for account fallback: rrStickyTargets map + clampStickyRoundRobinTargetLimit + getStickyRoundRobinStartIndex/recordStickyRoundRobinSuccess helpers wired into handleRoundRobinCombo, with sticky-eviction tied to rrCounters eviction). Cohesive routing logic in the combo handler; not a movable block. Structural shrink of combo.ts tracked in #3501.",
+ "_rebaseline_2026_06_15_3871_empty_pool": "PR #3871 own growth: combo.ts 5203->5204 (+1 = guard expandAutoComboCandidatePool against an empty candidatePool array — Array.isArray(pool) && pool.length > 0 so [] falls through to active-connection expansion instead of early-returning). One-line correctness fix; not extractable.",
"cap": 800,
"frozen": {
- "open-sse/config/providerRegistry.ts": 4703,
+ "open-sse/config/providerRegistry.ts": 4708,
"open-sse/executors/antigravity.ts": 1649,
"open-sse/executors/base.ts": 1218,
"open-sse/executors/chatgpt-web.ts": 2870,
@@ -30,14 +34,14 @@
"open-sse/services/batchProcessor.ts": 828,
"open-sse/services/browserBackedChat.ts": 850,
"open-sse/services/claudeCodeCompatible.ts": 1202,
- "open-sse/services/combo.ts": 5203,
+ "open-sse/services/combo.ts": 5277,
"open-sse/services/rateLimitManager.ts": 1017,
"open-sse/services/tokenRefresh.ts": 1997,
"open-sse/services/usage.ts": 3408,
"open-sse/translator/request/openai-to-gemini.ts": 844,
"open-sse/translator/response/openai-responses.ts": 878,
"open-sse/utils/cursorAgentProtobuf.ts": 1521,
- "open-sse/utils/stream.ts": 2710,
+
"src/app/(dashboard)/dashboard/HomePageClient.tsx": 1385,
"src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx": 1020,
"src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 2909,
@@ -99,13 +103,15 @@
"src/shared/components/RequestLoggerV2.tsx": 1282,
"src/shared/components/analytics/charts.tsx": 1558,
"src/shared/constants/cliTools.ts": 875,
- "src/shared/constants/pricing.ts": 1508,
+ "src/shared/constants/pricing.ts": 1529,
"src/shared/constants/providers.ts": 3147,
"src/shared/constants/sidebarVisibility.ts": 1100,
"src/shared/services/cliRuntime.ts": 1090,
"src/shared/validation/schemas.ts": 2523,
"src/sse/handlers/chat.ts": 1425,
- "src/sse/services/auth.ts": 2216
+ "src/sse/services/auth.ts": 2216,
+ "open-sse/utils/stream/streamCore.ts": 2216,
+ "open-sse/utils/stream.ts": 2584
},
"_rebaseline_2026_06_09": "Re-baseline consciente pre-release v3.8.19: 9 arquivos cresceram durante o ciclo (features mergeadas: RequestLoggerV2 +281 request-logger rework, stream +101, combo +73, chatCore +45, catalog +32 fable-5/catalog-flag, callLogs +4, accountFallback +2, usageHistory novo 840) + core.ts +7 (fix resetAllDbModuleState, PR 3536). A catraca segue valendo destes valores — proximo crescimento falha. Decisao: encolher (esp. RequestLoggerV2/chatCore) e a issue #3501 ficam para o ciclo seguinte.",
"_rebaseline_2026_06_11_phase1f": "Phase 1f (#3501): ProviderDetailPageClient.tsx 4948→4062 (-886 LOC); 3 novos hooks extraídos. useProviderConnections.ts=954 acima do cap=800 — justificado: extração direta do god-component (zero lógica nova), própria redução do cliente supera o custo. useProviderSettings.ts=263 e useProviderModels.ts=154 já abaixo do cap.",
diff --git a/quality-baseline.json b/config/quality/quality-baseline.json
similarity index 100%
rename from quality-baseline.json
rename to config/quality/quality-baseline.json
diff --git a/test-discovery-baseline.json b/config/quality/test-discovery-baseline.json
similarity index 100%
rename from test-discovery-baseline.json
rename to config/quality/test-discovery-baseline.json
diff --git a/docs/architecture/REPOSITORY_MAP.md b/docs/architecture/REPOSITORY_MAP.md
index bddfefd9d56..c5333921082 100644
--- a/docs/architecture/REPOSITORY_MAP.md
+++ b/docs/architecture/REPOSITORY_MAP.md
@@ -1,13 +1,13 @@
---
title: "Repository Map"
-version: 3.8.2
-lastUpdated: 2026-05-13
+version: 3.8.26
+lastUpdated: 2026-06-15
---
# Repository Map
> **One-line description for every directory and root file.**
-> Last updated: 2026-05-13 — OmniRoute v3.8.0
+> Last updated: 2026-06-15 — OmniRoute v3.8.26
>
> Use this map to navigate the codebase quickly. For deep dives, follow links to dedicated docs.
@@ -23,8 +23,13 @@ OmniRoute/
├── docs/ # Public documentation (you are here)
├── tests/ # All test suites (unit, integration, e2e, protocols-e2e)
├── public/ # Next.js static assets, PWA manifest, service worker, icons
-├── config/ # Static config files
+├── config/ # Static config + quality-gate state (i18n, payloadRules, quality/)
├── images/ # Marketing / README image assets
+├── @omniroute/ # Publishable companion packages (opencode-plugin, opencode-provider)
+├── skills/ # CLI/agent skill packs (cli-* + omni-* + config-codex-cli)
+├── examples/ # Sample plugins + omniroute-cmd-hello starter
+├── contrib/ # Community contributions (podman/)
+├── .source/ # Fumadocs source config (source.config.mjs + server/browser/dynamic)
├── .github/ # GitHub Actions workflows + issue templates + PR template
├── .husky/ # Git hooks (pre-commit, pre-push)
├── .claude/ # Claude Code slash commands (project-scoped)
@@ -34,6 +39,7 @@ OmniRoute/
├── _mono_repo/ # Historic subprojects (cloud, site, vscode-extension)
├── _references/ # Read-only reference clones from related OSS projects
├── _tasks/ # Per-release task tracking files (informal)
+├── .build/ .worktrees/ dist/ # local build / git-worktree / build-output scratch (gitignored)
├── .issues/ # Local issue cache (gitignored)
├── .playwright-mcp/ # Playwright MCP test artifacts
├── coverage/ # c8 coverage output (gitignored)
@@ -48,45 +54,63 @@ OmniRoute/
## Root files
-| File | Purpose |
-| ------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
-| **README.md** | Marketing landing page + quick start + feature matrix (see also `llm.txt`) |
-| **CHANGELOG.md** | Per-release changelog (auto-generated by `/version-bump-cc` skill) |
-| **LICENSE** | MIT license text |
-| **CLAUDE.md** | Project rules for Claude Code agents (hard rules, conventions, scenarios) |
-| **AGENTS.md** | Same as CLAUDE.md but for non-Claude AI agents (Codex, Cursor, etc.) |
-| **GEMINI.md** | Concise rules for Gemini-based agents (subset of CLAUDE.md) |
-| **CONTRIBUTING.md** | Contributor guide: setup, conventional commits, testing, PR flow |
-| **SECURITY.md** | Vulnerability reporting policy, supported versions, threat model |
-| **CODE_OF_CONDUCT.md** | Contributor Covenant — community behavior expectations |
-| **llm.txt** | Plain-text landing optimized for LLM crawlers (SEO for AI assistants) |
-| **Tuto_Qdrant.md** | Tutorial for enabling Qdrant vector memory — **integration currently dormant** (see banner; primary memory docs in `docs/frameworks/MEMORY.md`) |
-| **package.json** | npm manifest, scripts, dependencies, engines, c8 coverage gate |
-| **package-lock.json** | Locked dependency tree |
-| **tsconfig.json** | Root TypeScript config |
-| **tsconfig.typecheck-core.json** | Typecheck config for `src/` core |
-| **tsconfig.typecheck-noimplicit-core.json** | Strict (`noImplicitAny`) typecheck |
-| **tsconfig.tsbuildinfo** | TS incremental build cache (gitignored) |
-| **next.config.mjs** | Next.js 16 build configuration (standalone output) |
-| **next-env.d.ts** | Next.js auto-generated env types |
-| **eslint.config.mjs** | ESLint flat config (rules per project area) |
-| **prettier.config.mjs** | Prettier formatting rules |
-| **postcss.config.mjs** | PostCSS config for Tailwind/CSS pipeline |
-| **playwright.config.ts** | Playwright E2E test config |
-| **vitest.config.ts** | Vitest config (default suite) |
-| **vitest.mcp.config.ts** | Vitest config for MCP server / autoCombo / cache suites |
-| **sonar-project.properties** | SonarQube/SonarCloud config (code quality) |
-| **Dockerfile** | Multi-stage Docker build (builder → runner-base → runner-cli) |
-| **docker-compose.yml** | Dev compose with 4 profiles (base, cli, host, cliproxyapi) + redis sidecar |
-| **docker-compose.prod.yml** | Production compose (port 20130, redis, named volumes) |
-| **.dockerignore** | Files excluded from Docker context |
-| **fly.toml** | Fly.io deployment config (region `sin`, port 20128, /data volume) |
-| **.env.example** | Template env file (815 lines, auto-copied to `.env` on first install) |
-| **.gitignore** | Git ignore patterns |
-| **.npmignore** | npm publish exclusion list |
-| **.npmrc** | npm config (registry, lockfile policy) |
-| **.node-version** | Node version pin (used by nvm-compatible tools) |
-| **.nvmrc** | Node version pin for nvm |
+| File | Purpose |
+| ------------------------------------------- | ---------------------------------------------------------------------------------------- |
+| **README.md** | Marketing landing page + quick start + feature matrix (see also `llm.txt`) |
+| **CHANGELOG.md** | Per-release changelog (auto-generated by `/version-bump-cc` skill) |
+| **LICENSE** | MIT license text |
+| **CLAUDE.md** | Project rules for Claude Code agents (hard rules, conventions, scenarios) |
+| **AGENTS.md** | Same as CLAUDE.md but for non-Claude AI agents (Codex, Cursor, etc.) |
+| **GEMINI.md** | Concise rules for Gemini-based agents (subset of CLAUDE.md) |
+| **CONTRIBUTING.md** | Contributor guide: setup, conventional commits, testing, PR flow |
+| **SECURITY.md** | Vulnerability reporting policy, supported versions, threat model |
+| **CODE_OF_CONDUCT.md** | Contributor Covenant — community behavior expectations |
+| **llm.txt** | Plain-text landing optimized for LLM crawlers (SEO for AI assistants) |
+| **package.json** | npm manifest, scripts, dependencies, engines, c8 coverage gate |
+| **package-lock.json** | Locked dependency tree |
+| **tsconfig.json** | Root TypeScript config |
+| **tsconfig.typecheck-core.json** | Typecheck config for `src/` core |
+| **tsconfig.typecheck-noimplicit-core.json** | Strict (`noImplicitAny`) typecheck |
+| **tsconfig.tsbuildinfo** | TS incremental build cache (gitignored) |
+| **next.config.mjs** | Next.js 16 build configuration (standalone output) |
+| **next-env.d.ts** | Next.js auto-generated env types |
+| **eslint.config.mjs** | ESLint flat config (rules per project area) |
+| **prettier.config.mjs** | Prettier formatting rules |
+| **postcss.config.mjs** | PostCSS config for Tailwind/CSS pipeline |
+| **playwright.config.ts** | Playwright E2E test config |
+| **vitest.config.ts** | Vitest config (default suite) |
+| **vitest.mcp.config.ts** | Vitest config for MCP server / autoCombo / cache suites |
+| **sonar-project.properties** | SonarQube/SonarCloud config (code quality) |
+| **Dockerfile** | Multi-stage Docker build (builder → runner-base → runner-cli) |
+| **docker-compose.yml** | Dev compose with 4 profiles (base, cli, host, cliproxyapi) + redis sidecar |
+| **docker-compose.prod.yml** | Production compose (port 20130, redis, named volumes) |
+| **.dockerignore** | Files excluded from Docker context |
+| **fly.toml** | Fly.io deployment config (region `sin`, port 20128, /data volume) |
+| **.env.example** | Template env file (auto-copied to `.env` on first install) |
+| **.gitignore** | Git ignore patterns |
+| **.npmignore** | npm publish exclusion list |
+| **.npmrc** | npm config (registry, lockfile policy) |
+| **.node-version** | Node version pin (used by nvm-compatible tools) |
+| **.nvmrc** | Node version pin for nvm |
+| **eslint.complexity.config.mjs** | ESLint config for the complexity ratchet (`scripts/check/check-complexity.mjs --config`) |
+| **eslint.sonarjs.config.mjs** | ESLint config for SonarJS rules (cognitive complexity / duplication) |
+| **source.config.ts** | Fumadocs `defineDocs` source config (feeds `.source/`) |
+| **knip.json** | Knip config — unused files/exports/deps (feeds the dead-code gate) |
+| **stryker.conf.json** | Stryker mutation-testing config |
+| **.size-limit.json** | size-limit bundle budget config |
+| **semcheck.yaml** | semcheck (spec↔code drift) config |
+| **promptfooconfig.yaml** | promptfoo eval config |
+| **.gitleaks.toml** | gitleaks secret-scan ruleset |
+| **.zizmor.yml** | zizmor GitHub-Actions security-lint config |
+| **socket.yml** | Socket.dev supply-chain config |
+| **news.json** | In-app release-notes feed (read by `src/shared/utils/releaseNotes.ts`) |
+| **flake.nix** / **flake.lock** | Nix dev-shell definition + lock |
+| **.env** | Local secrets (gitignored — generated from `.env.example`) |
+
+> **Moved out of the root in v3.8.26 (declutter):**
+>
+> - **→ `config/quality/`:** `quality-baseline.json`, `complexity-baseline.json`, `duplication-baseline.json`, `file-size-baseline.json`, `test-discovery-baseline.json`, `dependency-allowlist.json`, `.license-allowlist.json`, and the generated `quality-metrics.json` (gitignored). See [`## config/`](#config--static-configs--quality-gate-state).
+> - **→ `docs/ops/`:** `DOCUMENTATION_AUDIT_REPORT.md`.
---
@@ -117,84 +141,84 @@ src/
### `src/app/` — App Router (Next.js 16)
-| Path | Purpose |
-| ---------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| `app/api/v1/` | Public OpenAI-compat API (~25 sub-routes: chat, completions, embeddings, files, batches, audio, images, videos, music, rerank, moderations, search, ws, agents, accounts, providers, etc.) |
-| `app/api/v1beta/` | Gemini-style API endpoints |
-| `app/api/playground/` | Playground Studio routes: `improve-prompt/` (POST — LLM prompt rewriter), `presets/` (GET list / POST create), `presets/[id]/` (GET / PUT / DELETE) — see `docs/frameworks/PLAYGROUND_STUDIO.md` |
-| `app/api/` (non-v1) | Management/admin routes (~60 directories: providers, combos, settings, mcp, a2a, evals, memory, skills, webhooks, compliance, resilience, monitoring, tunnels, cli-tools, etc.) |
-| `app/api/tools/agent-bridge/` | AgentBridge REST API — 12 routes (server control, agent state/DNS/mappings, bypass, cert, upstream-CA). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/AGENTBRIDGE.md §7`. |
-| `app/api/tools/traffic-inspector/` | Traffic Inspector REST + WS API — 16+ routes (requests, sessions, hosts, capture-modes, export, ws). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/TRAFFIC_INSPECTOR.md §8`. |
-| `app/a2a/` | A2A JSON-RPC 2.0 entry point (`POST /a2a`) |
-| `app/.well-known/agent.json/` | A2A Agent Card (discovery) |
-| `app/(dashboard)/dashboard/` | Dashboard UI pages (~35 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, activity, etc.) |
-| `app/(dashboard)/dashboard/search-tools/` | Search Tools Studio UI (3 tabs: Search/Scrape/Compare + SearchConceptCard + ProviderCatalog) — see `docs/frameworks/SEARCH_TOOLS_STUDIO.md` |
-| `app/(dashboard)/dashboard/` | Dashboard UI pages (~30 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, etc.) |
+| Path | Purpose |
+| ---------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| `app/api/v1/` | Public OpenAI-compat API (~25 sub-routes: chat, completions, embeddings, files, batches, audio, images, videos, music, rerank, moderations, search, ws, agents, accounts, providers, etc.) |
+| `app/api/v1beta/` | Gemini-style API endpoints |
+| `app/api/playground/` | Playground Studio routes: `improve-prompt/` (POST — LLM prompt rewriter), `presets/` (GET list / POST create), `presets/[id]/` (GET / PUT / DELETE) — see `docs/frameworks/PLAYGROUND_STUDIO.md` |
+| `app/api/` (non-v1) | Management/admin routes (~60 directories: providers, combos, settings, mcp, a2a, evals, memory, skills, webhooks, compliance, resilience, monitoring, tunnels, cli-tools, etc.) |
+| `app/api/tools/agent-bridge/` | AgentBridge REST API — 12 routes (server control, agent state/DNS/mappings, bypass, cert, upstream-CA). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/AGENTBRIDGE.md §7`. |
+| `app/api/tools/traffic-inspector/` | Traffic Inspector REST + WS API — 16+ routes (requests, sessions, hosts, capture-modes, export, ws). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/TRAFFIC_INSPECTOR.md §8`. |
+| `app/a2a/` | A2A JSON-RPC 2.0 entry point (`POST /a2a`) |
+| `app/.well-known/agent.json/` | A2A Agent Card (discovery) |
+| `app/(dashboard)/dashboard/` | Dashboard UI pages (~35 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, activity, etc.) |
+| `app/(dashboard)/dashboard/search-tools/` | Search Tools Studio UI (3 tabs: Search/Scrape/Compare + SearchConceptCard + ProviderCatalog) — see `docs/frameworks/SEARCH_TOOLS_STUDIO.md` |
+| `app/(dashboard)/dashboard/` | Dashboard UI pages (~30 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, etc.) |
| `app/(dashboard)/dashboard/memory/` | Memory Studio (plan 21): `page.tsx` (3-tab shell), `components/` (MemoryConceptCard, MemoryEngineStatus, EmbeddingSourceSelector, EditMemoryModal, RetrievePreview, QdrantConfigCard, RerankConfigCard), `components/tabs/` (MemoriesTab, PlaygroundTab, EngineTab), `hooks/` (useEngineStatus, useMemorySettings) |
-| `app/(dashboard)/dashboard/tools/agent-bridge/` | AgentBridge dashboard page — server card, 9 agent cards, setup wizard, model mapping, bypass list. i18n PT-BR + EN. See `docs/frameworks/AGENTBRIDGE.md`. |
-| `app/(dashboard)/dashboard/tools/traffic-inspector/` | Traffic Inspector dashboard page — DevTools split, 7 detail tabs, 4 capture mode toggles, session recorder, context colorization. i18n PT-BR + EN. See `docs/frameworks/TRAFFIC_INSPECTOR.md`. |
-| `app/(dashboard)/dashboard/activity/` | Activity feed page (Group B): `page.tsx` (server) + `ActivityFeedClient.tsx` + `components/{ActivityFeed,ActivityItem,DayHeader,EventTypeFilter}.tsx` — see `docs/architecture/MONITORING_SECTIONS.md` |
-| `app/(dashboard)/dashboard/costs/quota-share/` | Quota Sharing page (Group B): `QuotaSharePageClient.tsx` + `components/{PoolCard,DimensionBar,AllocationTable,BurnRateChart,QuotaConceptCard,CreatePoolModal,EditAllocationsModal}.tsx` + `hooks/{usePools,usePoolUsage,useLocalStoragePoolMigration}.ts` |
-| `app/(dashboard)/dashboard/costs/quota-share/plans/` | Provider plan config page (Group B): `page.tsx` + `ProviderPlanConfigClient.tsx` — quota dimensions per connection override |
-| `app/docs/` | Embedded documentation viewer (renders `docs/*.md`) |
-| `app/landing/` | Marketing landing page |
-| `app/login/`, `forgot-password/`, `forbidden/` | Auth-related pages |
-| `app/{400,401,403,408,429,500,502,503}/` | HTTP error pages |
-| `app/maintenance/`, `offline/`, `status/`, `privacy/`, `terms/`, `callback/` | Static/status pages |
-| `app/layout.tsx`, `page.tsx`, `manifest.ts`, `globals.css` | Root layout, home, PWA manifest, global CSS |
-| `app/error.tsx`, `global-error.tsx`, `not-found.tsx`, `loading.tsx` | Error boundaries |
+| `app/(dashboard)/dashboard/tools/agent-bridge/` | AgentBridge dashboard page — server card, 9 agent cards, setup wizard, model mapping, bypass list. i18n PT-BR + EN. See `docs/frameworks/AGENTBRIDGE.md`. |
+| `app/(dashboard)/dashboard/tools/traffic-inspector/` | Traffic Inspector dashboard page — DevTools split, 7 detail tabs, 4 capture mode toggles, session recorder, context colorization. i18n PT-BR + EN. See `docs/frameworks/TRAFFIC_INSPECTOR.md`. |
+| `app/(dashboard)/dashboard/activity/` | Activity feed page (Group B): `page.tsx` (server) + `ActivityFeedClient.tsx` + `components/{ActivityFeed,ActivityItem,DayHeader,EventTypeFilter}.tsx` — see `docs/architecture/MONITORING_SECTIONS.md` |
+| `app/(dashboard)/dashboard/costs/quota-share/` | Quota Sharing page (Group B): `QuotaSharePageClient.tsx` + `components/{PoolCard,DimensionBar,AllocationTable,BurnRateChart,QuotaConceptCard,CreatePoolModal,EditAllocationsModal}.tsx` + `hooks/{usePools,usePoolUsage,useLocalStoragePoolMigration}.ts` |
+| `app/(dashboard)/dashboard/costs/quota-share/plans/` | Provider plan config page (Group B): `page.tsx` + `ProviderPlanConfigClient.tsx` — quota dimensions per connection override |
+| `app/docs/` | Embedded documentation viewer (renders `docs/*.md`) |
+| `app/landing/` | Marketing landing page |
+| `app/login/`, `forgot-password/`, `forbidden/` | Auth-related pages |
+| `app/{400,401,403,408,429,500,502,503}/` | HTTP error pages |
+| `app/maintenance/`, `offline/`, `status/`, `privacy/`, `terms/`, `callback/` | Static/status pages |
+| `app/layout.tsx`, `page.tsx`, `manifest.ts`, `globals.css` | Root layout, home, PWA manifest, global CSS |
+| `app/error.tsx`, `global-error.tsx`, `not-found.tsx`, `loading.tsx` | Error boundaries |
### `src/lib/` — Core libraries (~50 modules)
-| Module | Purpose |
-| ---------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `a2a/` | A2A protocol task manager, skills (5), streaming |
-| `acp/` | CLI Agent Registry (local CLI discovery — see `docs/frameworks/AGENT_PROTOCOLS_GUIDE.md`) |
-| `api/` | Shared API helpers (`requireManagementAuth`, validation) |
-| `auth/` | Session, password hashing, token validation |
-| `batches/` | OpenAI Batches API handlers |
-| `catalog/` | Provider catalog Zod validation + capability resolution |
-| `cloudAgent/` | Cloud Agents (Codex Cloud, Devin, Jules) — see `docs/frameworks/CLOUD_AGENT.md` |
-| `combos/` | Combo resolution + reorder helpers |
-| `audit/` | Activity feed helpers: `highLevelActions.ts` (allowlist + `isHighLevelAction()`), `activityIcons.ts` (action → icon/verb map), `timeline.ts` (groupByDay/relativeTime) — see `docs/architecture/MONITORING_SECTIONS.md` |
-| `compliance/` | Audit log + provider audit — see `docs/security/COMPLIANCE.md` |
-| `compression/` | Compression engine glue (engines live in `open-sse/services/compression/`) |
-| `config/` | Runtime config helpers |
-| `db/` | 45+ domain DB modules + 55 migrations (always go through here for SQLite) |
+| Module | Purpose |
+| ---------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `a2a/` | A2A protocol task manager, skills (5), streaming |
+| `acp/` | CLI Agent Registry (local CLI discovery — see `docs/frameworks/AGENT_PROTOCOLS_GUIDE.md`) |
+| `api/` | Shared API helpers (`requireManagementAuth`, validation) |
+| `auth/` | Session, password hashing, token validation |
+| `batches/` | OpenAI Batches API handlers |
+| `catalog/` | Provider catalog Zod validation + capability resolution |
+| `cloudAgent/` | Cloud Agents (Codex Cloud, Devin, Jules) — see `docs/frameworks/CLOUD_AGENT.md` |
+| `combos/` | Combo resolution + reorder helpers |
+| `audit/` | Activity feed helpers: `highLevelActions.ts` (allowlist + `isHighLevelAction()`), `activityIcons.ts` (action → icon/verb map), `timeline.ts` (groupByDay/relativeTime) — see `docs/architecture/MONITORING_SECTIONS.md` |
+| `compliance/` | Audit log + provider audit — see `docs/security/COMPLIANCE.md` |
+| `compression/` | Compression engine glue (engines live in `open-sse/services/compression/`) |
+| `config/` | Runtime config helpers |
+| `db/` | 45+ domain DB modules + 55 migrations (always go through here for SQLite) |
| `quota/` | Quota Sharing Engine: `dimensions.ts` (types/Zod), `types.ts` (QuotaStore interface), `sqliteQuotaStore.ts`, `redisQuotaStore.ts`, `storeFactory.ts`, `fairShare.ts`, `burnRate.ts`, `planResolver.ts`, `planRegistry.ts`, `saturationSignals.ts`, `enforce.ts`, `spendRecorder.ts` — see `docs/routing/QUOTA_SHARE.md` |
-| `display/` | UI formatting helpers (cost, latency, etc.) |
-| `embeddings/` | Embeddings service helpers |
-| `env/` | Env variable parsing + validation |
-| `evals/` | Eval framework (suites, runner, runtime) — see `docs/frameworks/EVALS.md` |
-| `guardrails/` | PII masker, prompt injection, vision bridge — see `docs/security/GUARDRAILS.md` |
-| `jobs/` | Background jobs (cron-like) |
-| `memory/` | Conversational memory (SQLite FTS5 + sqlite-vec hybrid RRF + Qdrant tier 2) — see `docs/frameworks/MEMORY.md` |
-| `memory/embedding/` | Multi-source embedding layer: `index.ts` (resolver), `remote.ts`, `staticPotion.ts`, `transformersLocal.ts`, `cache.ts`, `types.ts` (plan 21) |
-| `memory/vectorStore.ts` | sqlite-vec v0.1.9 wrapper — KNN brute-force + hybrid RRF (FTS5 + vector, k=60). Lazy-init, degrades gracefully when sqlite-vec unavailable. (plan 21) |
-| `memory/reindex.ts` | `runReindexBatch()` — processes memories with `needs_reindex=1` in background; called by `POST /api/memory/reindex` and lazy-backfill path. (plan 21) |
-| `monitoring/` | Health checks, metrics emission |
-| `oauth/` | OAuth flows for 14 providers (claude, codex, antigravity, cursor, github, gemini, kimi-coding, kilocode, cline, qwen, kiro, qoder, gitlab-duo, windsurf) |
-| `plugins/` | Plugin registry |
-| `promptCache/` | Anthropic-style prompt cache breakpoints |
-| `skills/` | Skills framework (built-in + marketplace + SkillsSH) — see `docs/frameworks/SKILLS.md` |
-| `playground/` | Playground Studio shared helpers: `codeExport.ts` (curl/Python/TS generator), `promptImprover.ts` (meta-prompt builder), `streamMetrics.ts` (pure TTFT/TPS), `types.ts` (pricing table) — see `docs/frameworks/PLAYGROUND_STUDIO.md` |
-| `webhookDispatcher.ts` | HMAC webhook delivery — see `docs/frameworks/WEBHOOKS.md` |
-| `cloudflaredTunnel.ts`, `ngrokTunnel.ts` | Tunnel managers — see `docs/ops/TUNNELS_GUIDE.md` |
-| `oneproxySync.ts`, `oneproxyRotator.ts` | 1proxy free proxy marketplace — see `docs/ops/PROXY_GUIDE.md` |
-| `cloudSync.ts`, `initCloudSync.ts` | Optional cloud sync of state |
-| `localDb.ts` | Re-export barrel for db modules (no logic — re-exports only) |
-| `cacheLayer.ts`, `idempotencyLayer.ts` | Request caching + idempotency |
-| (~30 more top-level files) | Specialized helpers (logEnv, modelsDevSync, piiSanitizer, etc.) |
+| `display/` | UI formatting helpers (cost, latency, etc.) |
+| `embeddings/` | Embeddings service helpers |
+| `env/` | Env variable parsing + validation |
+| `evals/` | Eval framework (suites, runner, runtime) — see `docs/frameworks/EVALS.md` |
+| `guardrails/` | PII masker, prompt injection, vision bridge — see `docs/security/GUARDRAILS.md` |
+| `jobs/` | Background jobs (cron-like) |
+| `memory/` | Conversational memory (SQLite FTS5 + sqlite-vec hybrid RRF + Qdrant tier 2) — see `docs/frameworks/MEMORY.md` |
+| `memory/embedding/` | Multi-source embedding layer: `index.ts` (resolver), `remote.ts`, `staticPotion.ts`, `transformersLocal.ts`, `cache.ts`, `types.ts` (plan 21) |
+| `memory/vectorStore.ts` | sqlite-vec v0.1.9 wrapper — KNN brute-force + hybrid RRF (FTS5 + vector, k=60). Lazy-init, degrades gracefully when sqlite-vec unavailable. (plan 21) |
+| `memory/reindex.ts` | `runReindexBatch()` — processes memories with `needs_reindex=1` in background; called by `POST /api/memory/reindex` and lazy-backfill path. (plan 21) |
+| `monitoring/` | Health checks, metrics emission |
+| `oauth/` | OAuth flows for 14 providers (claude, codex, antigravity, cursor, github, gemini, kimi-coding, kilocode, cline, qwen, kiro, qoder, gitlab-duo, windsurf) |
+| `plugins/` | Plugin registry |
+| `promptCache/` | Anthropic-style prompt cache breakpoints |
+| `skills/` | Skills framework (built-in + marketplace + SkillsSH) — see `docs/frameworks/SKILLS.md` |
+| `playground/` | Playground Studio shared helpers: `codeExport.ts` (curl/Python/TS generator), `promptImprover.ts` (meta-prompt builder), `streamMetrics.ts` (pure TTFT/TPS), `types.ts` (pricing table) — see `docs/frameworks/PLAYGROUND_STUDIO.md` |
+| `webhookDispatcher.ts` | HMAC webhook delivery — see `docs/frameworks/WEBHOOKS.md` |
+| `cloudflaredTunnel.ts`, `ngrokTunnel.ts` | Tunnel managers — see `docs/ops/TUNNELS_GUIDE.md` |
+| `oneproxySync.ts`, `oneproxyRotator.ts` | 1proxy free proxy marketplace — see `docs/ops/PROXY_GUIDE.md` |
+| `cloudSync.ts`, `initCloudSync.ts` | Optional cloud sync of state |
+| `localDb.ts` | Re-export barrel for db modules (no logic — re-exports only) |
+| `cacheLayer.ts`, `idempotencyLayer.ts` | Request caching + idempotency |
+| (~30 more top-level files) | Specialized helpers (logEnv, modelsDevSync, piiSanitizer, etc.) |
### `src/db/` — Database (45+ modules + 55 migrations)
-| Subdir | Purpose |
-| ---------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `db/core.ts` | `getDbInstance()` singleton with WAL journaling |
-| `db/migrations/` | Versioned SQL files (idempotent, transactional). `073_memory_vec.sql` adds `memory_vec_meta` + `needs_reindex` column (plan 21). |
-| `db/playgroundPresets.ts` | CRUD module for Playground Studio presets (`listPlaygroundPresets`, `getPlaygroundPreset`, `createPlaygroundPreset`, `updatePlaygroundPreset`, `deletePlaygroundPreset`) |
-| `db/memoryVec.ts`| CRUD for `memory_vec_meta` (active_dim, embedding_signature, last_reset_at, vec_loaded) + `markMemoryNeedsReindex`, `getMemoryReindexQueue`, etc. (plan 21) |
-| `db/.ts` | One module per domain: providers, combos, apiKeys, users, sessions, usage, audit*log, webhooks, skills, memory_entries, cloud_agent_tasks, evals*\*, reasoning_cache, etc. |
+| Subdir | Purpose |
+| ------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `db/core.ts` | `getDbInstance()` singleton with WAL journaling |
+| `db/migrations/` | Versioned SQL files (idempotent, transactional). `073_memory_vec.sql` adds `memory_vec_meta` + `needs_reindex` column (plan 21). |
+| `db/playgroundPresets.ts` | CRUD module for Playground Studio presets (`listPlaygroundPresets`, `getPlaygroundPreset`, `createPlaygroundPreset`, `updatePlaygroundPreset`, `deletePlaygroundPreset`) |
+| `db/memoryVec.ts` | CRUD for `memory_vec_meta` (active_dim, embedding_signature, last_reset_at, vec_loaded) + `markMemoryNeedsReindex`, `getMemoryReindexQueue`, etc. (plan 21) |
+| `db/.ts` | One module per domain: providers, combos, apiKeys, users, sessions, usage, audit*log, webhooks, skills, memory_entries, cloud_agent_tasks, evals*\*, reasoning_cache, etc. |
### `src/domain/`
@@ -455,9 +479,24 @@ open-sse/
---
-## `config/` — Static Configs
-
-Shipped configuration templates and sample files (referenced by setup wizard).
+## `config/` — Static Configs + Quality-Gate State
+
+Shipped configuration templates plus the committed quality-gate baselines
+(moved here from the repo root in v3.8.26 to keep the root lean).
+
+| Path | Purpose |
+| --------------------------------------------- | -------------------------------------------------------------------------------- |
+| `config/i18n.json` | Locale list + metadata (canonical source for the 42-locale count) |
+| `config/i18n-schema.json` | JSON schema validating `i18n.json` |
+| `config/payloadRules.json` | Upstream payload sanitization rules |
+| `config/quality/quality-baseline.json` | Multi-metric ratchet baseline (`scripts/quality/check-quality-ratchet.mjs`) |
+| `config/quality/complexity-baseline.json` | Frozen ESLint-complexity baseline (`check-complexity.mjs`) |
+| `config/quality/duplication-baseline.json` | Frozen jscpd duplication baseline (`check-duplication.mjs`) |
+| `config/quality/file-size-baseline.json` | Frozen per-file size baseline (`check-file-size.mjs`) |
+| `config/quality/test-discovery-baseline.json` | Frozen orphan-test baseline (`check-test-discovery.mjs`) |
+| `config/quality/dependency-allowlist.json` | Approved dependencies allowlist (`check-deps.mjs`) |
+| `config/quality/.license-allowlist.json` | SPDX license allowlist (`check-licenses.mjs`) |
+| `config/quality/quality-metrics.json` | Ephemeral collected metrics (generated by `collect-metrics.mjs`; **gitignored**) |
---
@@ -484,15 +523,15 @@ Shipped configuration templates and sample files (referenced by setup wizard).
## `.claude/` — Claude Code Slash Commands
-| File | Purpose |
-| ----------------------------------------------------------------- | -------------------------------------------------- |
-| `commands/version-bump-cc.md` | `/version-bump-cc` — bump version + auto-changelog |
-| `commands/generate-release-cc.md` | `/generate-release-cc` — full release workflow |
-| `commands/deploy-vps-{local,akamai,both}-cc.md` | Deploy to VPS |
-| `commands/capture-release-evidences-cc.md` | Browser-record new features as WebP |
-| `commands/review-{prs,discussions}-cc.md` | Triage GitHub PRs/discussions |
-| `commands/{review-issues,implement-features}-cc.md` | Issue workflows |
-| `settings.local.json` | Per-project Claude Code settings |
+| File | Purpose |
+| --------------------------------------------------- | -------------------------------------------------- |
+| `commands/version-bump-cc.md` | `/version-bump-cc` — bump version + auto-changelog |
+| `commands/generate-release-cc.md` | `/generate-release-cc` — full release workflow |
+| `commands/deploy-vps-{local,akamai,both}-cc.md` | Deploy to VPS |
+| `commands/capture-release-evidences-cc.md` | Browser-record new features as WebP |
+| `commands/review-{prs,discussions}-cc.md` | Triage GitHub PRs/discussions |
+| `commands/{review-issues,implement-features}-cc.md` | Issue workflows |
+| `settings.local.json` | Per-project Claude Code settings |
---
diff --git a/docs/guides/FEATURES.md b/docs/guides/FEATURES.md
index df79c1d0ef5..0ae629121b2 100644
--- a/docs/guides/FEATURES.md
+++ b/docs/guides/FEATURES.md
@@ -53,6 +53,8 @@ The v3.7.x → v3.8.0 cycle added zero-config auto routing, new providers, OAuth
Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+OpenRouter connections can store a per-connection `preset` in Advanced Settings. When set, OmniRoute sends it as the OpenRouter top-level request field, for example `"preset": "email-copywriter"`, unless the client request already supplied its own `preset`.
+

---
diff --git a/docs/i18n/ar/CHANGELOG.md b/docs/i18n/ar/CHANGELOG.md
index d4c2175cbfe..d2c25a654a8 100644
--- a/docs/i18n/ar/CHANGELOG.md
+++ b/docs/i18n/ar/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/az/CHANGELOG.md b/docs/i18n/az/CHANGELOG.md
index 4fb4d1ee043..ac25be0c6dc 100644
--- a/docs/i18n/az/CHANGELOG.md
+++ b/docs/i18n/az/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/bg/CHANGELOG.md b/docs/i18n/bg/CHANGELOG.md
index 4fb4d1ee043..ac25be0c6dc 100644
--- a/docs/i18n/bg/CHANGELOG.md
+++ b/docs/i18n/bg/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/bn/CHANGELOG.md b/docs/i18n/bn/CHANGELOG.md
index 47b5e3cec42..0834d753702 100644
--- a/docs/i18n/bn/CHANGELOG.md
+++ b/docs/i18n/bn/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/cs/CHANGELOG.md b/docs/i18n/cs/CHANGELOG.md
index ee25f9424af..4e03f9602b0 100644
--- a/docs/i18n/cs/CHANGELOG.md
+++ b/docs/i18n/cs/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/da/CHANGELOG.md b/docs/i18n/da/CHANGELOG.md
index 8f5040c48c1..fc5df4cbbaa 100644
--- a/docs/i18n/da/CHANGELOG.md
+++ b/docs/i18n/da/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/de/CHANGELOG.md b/docs/i18n/de/CHANGELOG.md
index b680e3dc980..1203fafe426 100644
--- a/docs/i18n/de/CHANGELOG.md
+++ b/docs/i18n/de/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/es/CHANGELOG.md b/docs/i18n/es/CHANGELOG.md
index 85dd13446cc..1fe9641e4fd 100644
--- a/docs/i18n/es/CHANGELOG.md
+++ b/docs/i18n/es/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/fa/CHANGELOG.md b/docs/i18n/fa/CHANGELOG.md
index 3060c2c78cf..b79d3e6d715 100644
--- a/docs/i18n/fa/CHANGELOG.md
+++ b/docs/i18n/fa/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/fi/CHANGELOG.md b/docs/i18n/fi/CHANGELOG.md
index d48ee8a418d..a012ba3fdbd 100644
--- a/docs/i18n/fi/CHANGELOG.md
+++ b/docs/i18n/fi/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/fr/CHANGELOG.md b/docs/i18n/fr/CHANGELOG.md
index 33e1682eb77..4ebd4e3b117 100644
--- a/docs/i18n/fr/CHANGELOG.md
+++ b/docs/i18n/fr/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/gu/CHANGELOG.md b/docs/i18n/gu/CHANGELOG.md
index 34c5b92c535..26119eb3b95 100644
--- a/docs/i18n/gu/CHANGELOG.md
+++ b/docs/i18n/gu/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/he/CHANGELOG.md b/docs/i18n/he/CHANGELOG.md
index dfb651c7e3b..cec30fe21a4 100644
--- a/docs/i18n/he/CHANGELOG.md
+++ b/docs/i18n/he/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/hi/CHANGELOG.md b/docs/i18n/hi/CHANGELOG.md
index 88193ed52d9..63b343e7412 100644
--- a/docs/i18n/hi/CHANGELOG.md
+++ b/docs/i18n/hi/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/hu/CHANGELOG.md b/docs/i18n/hu/CHANGELOG.md
index d8cba0e4c4f..703cc401fe1 100644
--- a/docs/i18n/hu/CHANGELOG.md
+++ b/docs/i18n/hu/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/id/CHANGELOG.md b/docs/i18n/id/CHANGELOG.md
index e89b56ac968..ccd63ed6d0c 100644
--- a/docs/i18n/id/CHANGELOG.md
+++ b/docs/i18n/id/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/in/CHANGELOG.md b/docs/i18n/in/CHANGELOG.md
index d8d45ea3543..fa54bd1fca5 100644
--- a/docs/i18n/in/CHANGELOG.md
+++ b/docs/i18n/in/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/it/CHANGELOG.md b/docs/i18n/it/CHANGELOG.md
index d592c0054a9..51c2d0fc795 100644
--- a/docs/i18n/it/CHANGELOG.md
+++ b/docs/i18n/it/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ja/CHANGELOG.md b/docs/i18n/ja/CHANGELOG.md
index 1a368244902..7c8b2f63175 100644
--- a/docs/i18n/ja/CHANGELOG.md
+++ b/docs/i18n/ja/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ko/CHANGELOG.md b/docs/i18n/ko/CHANGELOG.md
index 05d5e753db4..295d7b8e55e 100644
--- a/docs/i18n/ko/CHANGELOG.md
+++ b/docs/i18n/ko/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/mr/CHANGELOG.md b/docs/i18n/mr/CHANGELOG.md
index 19564c8d839..8f00e93e736 100644
--- a/docs/i18n/mr/CHANGELOG.md
+++ b/docs/i18n/mr/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ms/CHANGELOG.md b/docs/i18n/ms/CHANGELOG.md
index b4c645449f8..8a1c635018c 100644
--- a/docs/i18n/ms/CHANGELOG.md
+++ b/docs/i18n/ms/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/nl/CHANGELOG.md b/docs/i18n/nl/CHANGELOG.md
index 430ffc4ce6e..7cfe6406d96 100644
--- a/docs/i18n/nl/CHANGELOG.md
+++ b/docs/i18n/nl/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/no/CHANGELOG.md b/docs/i18n/no/CHANGELOG.md
index b14d9e9b9dd..4de0e1bf8da 100644
--- a/docs/i18n/no/CHANGELOG.md
+++ b/docs/i18n/no/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/phi/CHANGELOG.md b/docs/i18n/phi/CHANGELOG.md
index 7a69e46cd1e..2d0fd80b042 100644
--- a/docs/i18n/phi/CHANGELOG.md
+++ b/docs/i18n/phi/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/pl/CHANGELOG.md b/docs/i18n/pl/CHANGELOG.md
index 5a2b6c0af13..a6c0c4638c7 100644
--- a/docs/i18n/pl/CHANGELOG.md
+++ b/docs/i18n/pl/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/pt-BR/CHANGELOG.md b/docs/i18n/pt-BR/CHANGELOG.md
index 5db6d9b189e..2b69faea25f 100644
--- a/docs/i18n/pt-BR/CHANGELOG.md
+++ b/docs/i18n/pt-BR/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/pt/CHANGELOG.md b/docs/i18n/pt/CHANGELOG.md
index 364cc57c2e0..08bbf3db5ea 100644
--- a/docs/i18n/pt/CHANGELOG.md
+++ b/docs/i18n/pt/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ro/CHANGELOG.md b/docs/i18n/ro/CHANGELOG.md
index de3c1adc61b..a331e42c85a 100644
--- a/docs/i18n/ro/CHANGELOG.md
+++ b/docs/i18n/ro/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ru/CHANGELOG.md b/docs/i18n/ru/CHANGELOG.md
index e13ba3fb7a2..c2fa33e8b7b 100644
--- a/docs/i18n/ru/CHANGELOG.md
+++ b/docs/i18n/ru/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/sk/CHANGELOG.md b/docs/i18n/sk/CHANGELOG.md
index 40e3e3ae35b..74f75ba62cf 100644
--- a/docs/i18n/sk/CHANGELOG.md
+++ b/docs/i18n/sk/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/sv/CHANGELOG.md b/docs/i18n/sv/CHANGELOG.md
index 79e47d0c43c..c96db82ead4 100644
--- a/docs/i18n/sv/CHANGELOG.md
+++ b/docs/i18n/sv/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/sw/CHANGELOG.md b/docs/i18n/sw/CHANGELOG.md
index e0d05a0cb40..d9258be4c99 100644
--- a/docs/i18n/sw/CHANGELOG.md
+++ b/docs/i18n/sw/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ta/CHANGELOG.md b/docs/i18n/ta/CHANGELOG.md
index 60eb1fc2ce3..2a79a5512e4 100644
--- a/docs/i18n/ta/CHANGELOG.md
+++ b/docs/i18n/ta/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/te/CHANGELOG.md b/docs/i18n/te/CHANGELOG.md
index 78338abb5b1..62c87eba818 100644
--- a/docs/i18n/te/CHANGELOG.md
+++ b/docs/i18n/te/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/th/CHANGELOG.md b/docs/i18n/th/CHANGELOG.md
index c6aeb022bd9..ab7c3e44d91 100644
--- a/docs/i18n/th/CHANGELOG.md
+++ b/docs/i18n/th/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/tr/CHANGELOG.md b/docs/i18n/tr/CHANGELOG.md
index f39ee234763..67133c16578 100644
--- a/docs/i18n/tr/CHANGELOG.md
+++ b/docs/i18n/tr/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/uk-UA/CHANGELOG.md b/docs/i18n/uk-UA/CHANGELOG.md
index 258a01e20cb..9afca5cbc63 100644
--- a/docs/i18n/uk-UA/CHANGELOG.md
+++ b/docs/i18n/uk-UA/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/ur/CHANGELOG.md b/docs/i18n/ur/CHANGELOG.md
index c6834448a80..f740d738b2f 100644
--- a/docs/i18n/ur/CHANGELOG.md
+++ b/docs/i18n/ur/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/vi/CHANGELOG.md b/docs/i18n/vi/CHANGELOG.md
index ded47a11cfb..a0eb43f2a79 100644
--- a/docs/i18n/vi/CHANGELOG.md
+++ b/docs/i18n/vi/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/i18n/zh-CN/CHANGELOG.md b/docs/i18n/zh-CN/CHANGELOG.md
index 7dfd95db1eb..a2b4487077d 100644
--- a/docs/i18n/zh-CN/CHANGELOG.md
+++ b/docs/i18n/zh-CN/CHANGELOG.md
@@ -6,6 +6,14 @@
## [3.8.25] — 2026-06-14
+## [3.8.26] — TBD
+
+_See English CHANGELOG for v3.8.26 details._
+
+---
+
+## [3.8.25] — 2026-06-14
+
### ✨ New Features
- **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848))
diff --git a/docs/ops/DOCUMENTATION_AUDIT_REPORT.md b/docs/ops/DOCUMENTATION_AUDIT_REPORT.md
new file mode 100644
index 00000000000..767c97f3e24
--- /dev/null
+++ b/docs/ops/DOCUMENTATION_AUDIT_REPORT.md
@@ -0,0 +1,235 @@
+# OmniRoute — Relatório de Auditoria de Documentação & Plano de Sincronização
+
+> **Versão do projeto:** 3.8.24 · **Data:** 2026-06-13 · **Status:** FASE 1 (pesquisa/organização) — execução **pendente de confirmação**
+> **Escopo:** docs raiz · `/docs` · site `:20128/docs` (Fumadocs) · Wiki GitHub · i18n (42 locales) · CI de docs/i18n
+
+---
+
+## 0. TL;DR
+
+1. **As contagens estão dessincronizadas entre 5 fontes diferentes** (código, README, AGENTS.md, site, Wiki). O caso mais grave: **providers** aparece como `177` (README) / `232` (AGENTS) / `223` (gerador, **correto**) / `212+` (Wiki) / `160+` (CLAUDE.md).
+2. **Os gates de CI atuais NÃO validam os números mais visíveis** (provider count, free count, test count, locale count). Por isso o drift passou despercebido.
+3. **A Wiki do GitHub está órfã**: 995 páginas, **sem automação de sync**, último update genérico, números muito antigos (`14 strategies`, `37 MCP tools`, `212+ providers`).
+4. **O pipeline i18n nunca rodou para os docs**: `.i18n-state.json` não existe → drift check não tem baseline.
+5. **~10–12 funcionalidades recentes não têm documentação** (Plugin Marketplace, Free Provider Rankings/Arena ELO, IPv6 egress, Feature Flags, Notion/Obsidian context, etc.).
+
+---
+
+## 1. Números canônicos REAIS (a fonte de verdade de cada um)
+
+| Métrica | **Valor real** | Fonte de verdade (como medir) | README | AGENTS.md | CLAUDE.md | docs/README.md | Wiki Home | Site |
+| ------------------------------------- | ---------------------------------------------------- | --------------------------------------------------------------------------------- | -------------- | --------- | ---------- | --------------------------- | --------- | ----------------- | --- |
+| **Providers (total)** | **223** | `scripts/docs/gen-provider-reference.ts` → `PROVIDER_REFERENCE.md` ("unique IDs") | ❌ 177 | ❌ 232 | ❌ "160+" | (n/a) | ❌ 212+ | via gerador |
+| **Providers c/ free tier** | **103** `hasFree:true` / **98** pesquisados c/ quota | `grep hasFree:true providers.ts` / `FREE_TIERS.md` | ⚠️ "50+" | — | — | — | ⚠️ "50+" | — |
+| **Free forever** | **11** (a revalidar) | README claim — sem fonte programática | "11" | — | — | — | — | — |
+| **Test files (unit)** | **1.574** | `find tests/unit -name '*.test.ts'` | — | — | — | — | — | — |
+| **Test files (integration)** | **76** | `find tests/integration` | — | — | — | — | — | — |
+| **Test files (total)** | **~1.660** (+46 em src/open-sse) | find global | — | — | — | — | — | — |
+| **Test cases (aprox)** | **~16.000** | `grep -E '(test | it)\(' tests/` | — | — | — | — | — | — |
+| **`unit/` test files (CONTRIBUTING)** | **1.574** | — | — | — | — | ❌ **CONTRIBUTING diz 122** | — | — |
+| **API endpoints (route.ts)** | **502** | `find src/app/api -name route.ts` | — | — | — | — | — | — |
+| **Endpoints `/v1` (OpenAI-compat)** | **75** | `find src/app/api/v1 -name route.ts` | — | — | — | — | — | — |
+| **MCP tools** | **87** (33 base + módulos) | `schemas/tools.ts` = 33 base; +memory/skill/notion/obsidian/gamification/plugin | ✅ 87 | ✅ 87 | ✅ 87 | — | ❌ 37 | — |
+| **MCP scopes** | **30** (16 base em tools.ts) | `scopeEnforcement.ts` + módulos | — | ✅ 30 | ✅ 30 | — | — | — |
+| **Routing strategies** | **15** | `open-sse/services/combo.ts` (gate valida) | ✅ 15 | ✅ 15 | ✅ 15 | ❌ 14 | ❌ 14 | — |
+| **Auto-combo scoring factors** | **9** (label) / engine multifator | `AUTO-COMBO.md` | "9" | "12" | "9-factor" | ❌ "9-factor" | — | — |
+| **i18n locales** | **42** (+en = 43) | `config/i18n.json` | — | ✅ 42 | — | ❌ 40 | ❌ "40+" | ✅ 40 (LANGUAGES) |
+| **Executors** | **60** | gate valida ✓ | — | ✅ | — | — | — | — |
+| **A2A skills** | **6** | gate valida ✓ | ✅ | ✅ | ✅ | — | — | — |
+| **Cloud agents** | **3** | gate valida ✓ | ✅ | — | — | — | — | — |
+| **OAuth flows / providers** | **16** flows / **19** providers OAuth | gate (16) vs `PROVIDER_REFERENCE` (19) | — | — | — | — | — | — |
+| **DB modules / migrations** | **83 / 97** | gate/CLAUDE ✓ | — | ✅ | ✅ | — | — | — |
+
+> ⚠️ **Inconsistências de número que precisam de decisão de produto (não só correção mecânica):**
+>
+> - **Free count:** `hasFree:true` = 103, mas inclui créditos-de-cadastro one-time. `FREE_TIERS.md` documenta 98 pesquisados (≈63 recorrentes, 29 signup-only, 6 descontinuados). O "50+/11 forever" é conservador e defensável — **decidir o headline canônico**.
+> - **Auto-combo "9-factor" vs "12":** README diz 9, AGENTS diz 12. Precisa alinhar à contagem real em `AUTO-COMBO.md`.
+
+---
+
+## 2. Defasagens por fonte de documentação
+
+### 2.1 Documentos da raiz
+
+| Arquivo | Problema | Ação |
+| -------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- |
+| `README.md` | `177 providers` em ~9 lugares (linhas 9, 22, 23, 62, 139, 143, 266, 322, 326, 736) + badges + âncora `#-177-ai-providers--50-free` | Corrigir para **223**; revisar badge/âncora; revalidar "50+/11 forever" |
+| `AGENTS.md` | `232 provider entries` (linha 6) + live counts `providers 232` (linha 11) | Corrigir para **223** |
+| `CLAUDE.md` | `"160+"` providers (linha ~40) | Corrigir para **223** |
+| `CONTRIBUTING.md` | `unit/ (122 test files)` (linha 255) — defasado em >1.400 | Corrigir para **1.574** (ou texto dinâmico) |
+| `CHANGELOG.md` | OK (v3.8.24 correto) | — |
+| `SECURITY.md` / `CODE_OF_CONDUCT.md` / `GEMINI.md` | Genéricos, OK | — |
+
+### 2.2 `/docs`
+
+| Arquivo | Problema |
+| ---------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| `docs/README.md` (índice) | `9-factor scoring, 14 strategies` (linha 81 → deveria ser 15); `40 locales` (linha 121 → 42) |
+| `docs/guides/I18N.md` | "supports 30 languages" (real: 42); `lastUpdated 2026-05-13` |
+| `docs/frameworks/AGENT-SKILLS.md, AGENTBRIDGE.md` | `lastUpdated 2026-05-28` (v3.8.6) |
+| `docs/frameworks/WEBHOOKS.md`, `docs/guides/PWA_GUIDE.md` | `lastUpdated 2026-05-13` (v3.8.0) |
+| `docs/frameworks/SEARCH_TOOLS_STUDIO.md`, `PLAYGROUND_STUDIO.md` | `lastUpdated 2026-05-30` |
+| `docs/guides/TROUBLESHOOTING.md, FEATURES.md, UNINSTALL.md`, `docs/reference/API_REFERENCE.md` | refs a versões antigas (v3.5.x–v3.7.x) — verificar se históricas (ok) ou stale |
+| Órfãos do site (existem em `/docs` mas fora do `meta.json`) | `routing/QUOTA_SHARE.md`, `guides/CODEX-CLI-CONFIGURATION.md`, `security/SOCKET_DEV_FINDINGS.md`, `compression/EXTENDING_COMPRESSION.md` — acessíveis por URL mas **não na sidebar** |
+| Raiz `/docs` nunca no site | `AGENTROUTER.md`, `PROVIDERS.md`, `DOCUMENTATION_OVERHAUL_PLAN.md`, `SUBMIT_PR.md`, `fix-opencode-context.md` |
+
+### 2.3 Site `:20128/docs` (Fumadocs)
+
+- **Como funciona:** `docs//*.md` → `source.config.ts` (globs) → `.source/server.ts` (gerado) → `src/lib/source.ts` → `src/app/docs/layout.tsx` (sidebar = `pageTree` dos `meta.json`) → `[...slug]/page.tsx`. **60 docs em inglês** entram no site.
+- **Navegação curada por `meta.json`** → arquivo novo em `/docs` **não aparece** até ser adicionado manualmente ao `meta.json` da seção. Hoje há 4 arquivos importados mas fora da sidebar (acima).
+- **i18n no site:** `[...slug]/page.tsx` lê cookie `NEXT_LOCALE`; se ≠ en, tenta `docs/i18n//docs//.md` via `marked.parse()`, com fallback para o MDX inglês. Seletor: `LanguageSelector.tsx` (40 idiomas em `LANGUAGES`).
+- **API Explorer:** `openapi.generated.ts` é gerado por `scripts/docs/gen-openapi-module.mjs` a partir de `docs/reference/openapi.yaml` no `prebuild:docs`.
+- **Riscos de drift:** (a) `meta.json` manual; (b) traduções não atualizam quando o inglês muda; (c) `openapi.yaml` precisa de regen; (d) `LANGUAGES` no app diz 40, config diz 42 → **divergência app vs config**.
+
+### 2.4 Wiki do GitHub (`/wiki`) — **mais defasada de todas**
+
+- **995 páginas** (`60 docs Title-Case` + `935 i18n`), `Home.md`, `_Sidebar.md`, `_Footer.md`.
+- **Sem nenhum script/automação de sync** no repo (`grep wiki` em `scripts/`, `.github/`, `package.json` = vazio). Foi populada uma vez, manualmente.
+- **Números muito antigos no `Home.md`:** `212+ providers`, `14 Routing Strategies`, `MCP Server 37 tools`, `40+ Languages`.
+- **Conclusão:** a Wiki não tem "fonte de verdade" — precisa virar **espelho automatizado** de `/docs` (ver Plano §6, Fase 4).
+
+### 2.5 i18n (42 locales)
+
+- **Fonte de verdade dos locales:** `config/i18n.json` → **42** (+en=43). Documentação diz 30 (`I18N.md`) e 40 (`docs/README.md`, `LANGUAGES` no app) — **3 números diferentes**.
+- **Subset espelhado por idioma:** ~26–27 arquivos (7 raiz: README/CONTRIBUTING/CLAUDE/GEMINI/AGENTS/SECURITY/CODE_OF_CONDUCT + llm.txt/CHANGELOG copiados + ~19 docs em architecture/frameworks/guides/ops/reference/routing).
+- **`.i18n-state.json` não existe** → `i18n:check` (drift) não tem baseline; tradução de docs nunca foi executada pelo pipeline novo.
+- **Duplicações/legados:** `pt` vs `pt-BR` (ambos traduzem os mesmos arquivos); `id` vs `in` (Indonésio — `in` é legado ISO 639-3, deveria deprecar).
+- **CLI locales incompletos:** `bn, gu, he, mr, ms, phi, in` = 3 bytes (vazios).
+- **Motor:** `run-translation.mjs` usa endpoint OpenAI-compat via env `OMNIROUTE_TRANSLATION_*` (LLM); scripts Python (`i18n_autotranslate.py`, `generate-multilang.mjs`) marcados deprecated.
+
+---
+
+## 3. Gaps de funcionalidades (features sem doc) — com curadoria
+
+> Curadoria aplicada: o agente de exploração marcou 57% dos módulos como "não documentados", mas muitos (`config`, `runtime`, `middleware`, `images`, `catalog`, `system`, `display`, `events`, `embeddings`) são **internos** e não merecem doc dedicado. Lista abaixo filtrada para o que é **voltado ao usuário/operador**.
+
+### P0 — features novas visíveis ao usuário, sem doc
+
+| Feature | Onde no código | PR | Doc sugerido |
+| ------------------------------------------------------ | -------------------------------------------------------------------------- | ------------ | ---------------------------------------------------------- |
+| **Plugin Marketplace** (customizável + SSRF hardening) | `src/app/api/plugins/marketplace/` | #3656, #3774 | `docs/frameworks/PLUGIN_MARKETPLACE.md` |
+| **Free Provider Rankings (Arena ELO)** | `src/app/api/free-provider-rankings/`, `/dashboard/free-provider-rankings` | #3799 | `docs/guides/FREE_PROVIDER_RANKINGS.md` |
+| **IPv6 egress family selector** (auto/ipv4/ipv6) | proxy/egress + UI form | #3777 | `docs/security/EGRESS_POLICY.md` (ou seção em PROXY_GUIDE) |
+| **Feature Flags page (runtime + emergency fallback)** | `/dashboard` feature-flags | #3752, #3741 | `docs/reference/FEATURE_FLAGS.md` |
+
+### P1 — integrações/frameworks sem doc
+
+| Feature | Onde | Doc sugerido |
+| -------------------------------------- | ----------------------------------- | ------------------------------------- |
+| **Notion context source** | `src/lib/notion/` (+6 MCP tools) | `docs/frameworks/NOTION_CONTEXT.md` |
+| **Obsidian context source** | `src/lib/obsidian/` (+22 MCP tools) | `docs/frameworks/OBSIDIAN_CONTEXT.md` |
+| **Quota-shared routing audit** (#3779) | combo + quota | seção em `AUTO-COMBO.md` |
+| **Model lockout / success-decay** | `RESILIENCE_GUIDE.md` desatualizado | atualizar `RESILIENCE_GUIDE.md` |
+| **Cost/Spend tracking** | `/dashboard/costs` | `docs/guides/COST_TRACKING.md` |
+
+### P2 — sub-documentados
+
+Traffic Inspector, Search Tools Studio (raso), Prompt Caching, Credential Health, Background Jobs, Database Migrations guide.
+
+> **Validação obrigatória na execução:** cada item acima será confirmado no código (trust-but-verify) antes de escrever doc — não documentar feature que não exista/esteja como descrita.
+
+---
+
+## 4. CI de docs/i18n — coberto vs lacunas
+
+### Coberto (hard gates)
+
+`check:docs-sync` (version package↔openapi↔CHANGELOG + mirrors i18n) · `check:env-doc-sync` (env code↔.env.example↔ENVIRONMENT.md) · `check:docs-symbols` (anti-alucinação rota) · `check:openapi-routes` · `check:cli-i18n` · `check-ui-keys-coverage` (floor 65%).
+
+### Parcial / advisory
+
+`check:docs-counts` (**soft, não no CI principal** — e **não cobre providers/free/tests/locales**) · `check:doc-links` (só internos) · `check:fabricated-docs` (soft) · `check-translation-drift` (`--warn`, não bloqueia) · `validate_translation.py` (matrix `continue-on-error`).
+
+### Lacunas (sem gate algum)
+
+1. **Provider count / free count / test count / locale count** — os números mais visíveis **não são validados**. → causa-raiz de todo o drift atual.
+2. **Validação MDX/Fumadocs** — quebra de sintaxe só aparece no deploy.
+3. **Lint de prosa (Vale/markdownlint)** — sem checagem de estilo/ortografia.
+4. **Links externos** — `check:doc-links` ignora http(s); URLs mortas passam.
+5. **Imagens/screenshots/diagramas órfãos.**
+6. **Drift de tradução não é blocking.**
+7. **`meta.json` ↔ `/docs`** — arquivo novo fora da sidebar não é detectado.
+8. **Wiki sync** — inexistente.
+9. **`config/i18n.json` (42) vs `LANGUAGES` app (40)** — sem gate de consistência.
+
+---
+
+## 5. Boas práticas 2026 (pesquisa web) aplicáveis
+
+| Prática | Ferramenta | Aplicação no OmniRoute |
+| ----------------------------------- | -------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------- |
+| **Lint de prosa em CI** | **Vale** (Google/Microsoft style) + **markdownlint** | Novo job `docs-lint` (warning-first p/ não travar) |
+| **Severidade graduada** | error = link quebrado / code-fence / alt-text faltando; warning = voz passiva / estilo | Configurar `.vale.ini` + `.markdownlint.json` |
+| **Feedback local rápido** | pre-commit com markdownlint/Vale (<2s) | Adicionar ao husky lint-staged p/ `*.md` |
+| **Accept-list de vocabulário** | `Vale accept.txt` | Evitar ruído com termos do projeto (OmniRoute, combo, etc.) |
+| **Link checker robusto** | **lychee** (Rust, externos + internos, cache) | Job semanal/cron + flag opcional no doc-links |
+| **Wiki como espelho automatizado** | **`Andrew-Chen-Wang/github-wiki-action`** ou `wiki-sync` | Workflow que espelha `/docs` → `.wiki.git` em push to main |
+| **Tradução LLM roteada por tarefa** | docs técnicos → GPT-5.x; nuance → Claude; bulk → modelo barato | Já temos roteamento próprio — usar `cx/gpt-5.4-mini` p/ docs (config existente) |
+| **Translation memory + glossário** | reduz drift e protege termos | Adotar glossário/accept-list compartilhado UI+docs |
+| **TMS via MCP** | Crowdin/Lokalise/Tolgee/SimpleLocalize têm MCP oficial | Opcional futuro; hoje pipeline próprio já cobre |
+| **Gerar docs no CI** | docs sempre refletem o código | Elevar gerador de provider/openapi + count-guard a gate |
+
+**Fontes:** [Fern — Docs Linting Guide (jan/2026)](https://buildwithfern.com/post/docs-linting-guide) · [Netlify — Docs Linting in CI/CD](https://www.netlify.com/blog/a-key-to-high-quality-documentation-docs-linting-in-ci-cd/) · [GitLab Docs — Documentation testing](https://docs.gitlab.com/development/documentation/testing/) · [Lokalise — Best LLM for translation 2026](https://lokalise.com/blog/what-is-the-best-llm-for-translation/) · [Crowdin — AI Localization 2026](https://crowdin.com/blog/ai-localization) · [Andrew-Chen-Wang/github-wiki-action](https://github.com/Andrew-Chen-Wang/github-wiki-action) · [OneUptime — Generate Docs with GitHub Actions](https://oneuptime.com/blog/post/2026-01-27-generate-documentation-github-actions/view)
+
+---
+
+## 6. PLANO DE MELHORIAS & SINCRONIZAÇÃO (execução pós-confirmação)
+
+### Fase A — Números canônicos (correção mecânica de alto impacto)
+
+1. Regenerar `PROVIDER_REFERENCE.md` (`gen-provider-reference.ts`) e fixar **223** como fonte.
+2. Corrigir **providers** em: README (9 ocorrências + badge + âncora), AGENTS.md, CLAUDE.md, Wiki Home → **223**.
+3. Corrigir **tests** em CONTRIBUTING.md (122 → 1.574) — ou tornar texto dinâmico.
+4. Corrigir **strategies** (14 → 15) e **locales** (30/40 → 42) em `docs/README.md`, `I18N.md`, Wiki Home, e `LANGUAGES` do app.
+5. Corrigir **MCP tools** na Wiki (37 → 87) e alinhar **auto-combo factors** (9 vs 12 → valor real).
+6. Decidir headline **free** (50+/11 vs 98/103) e aplicar uniformemente.
+
+### Fase B — Sincronizar /docs + README (estrutural)
+
+7. Atualizar `lastUpdated`/versão dos docs defasados (AGENT-SKILLS, WEBHOOKS, PWA, I18N, SEARCH/PLAYGROUND_STUDIO).
+8. Adicionar os 4 arquivos órfãos ao `meta.json` (ou removê-los conscientemente).
+9. Avaliar README: adicionar/atualizar seções (Quick Start, tabela de números, links p/ novos docs).
+
+### Fase C — Novos documentos (gaps de features, P0→P1)
+
+10. Criar P0: PLUGIN_MARKETPLACE, FREE_PROVIDER_RANKINGS, EGRESS_POLICY, FEATURE_FLAGS.
+11. Atualizar RESILIENCE_GUIDE (model lockout) e AUTO-COMBO (quota-shared).
+12. Criar P1: NOTION_CONTEXT, OBSIDIAN_CONTEXT, COST_TRACKING (conforme confirmação).
+
+### Fase D — i18n
+
+13. Bootstrapar `.i18n-state.json` (`i18n:run --dry-run`) e rodar tradução dos docs corrigidos.
+14. Reconciliar `config/i18n.json` (42) ↔ `LANGUAGES` app (40); decidir sobre `in` (deprecar) e `pt`/`pt-BR`.
+15. Atualizar I18N.md com o processo real e contagem 42.
+
+### Fase E — Site `:20128/docs`
+
+16. Regenerar `openapi.generated.ts` e validar API Explorer.
+17. Garantir que os novos docs entram no `meta.json` e renderizam (verificação visual via browser).
+
+### Fase F — Wiki GitHub (automatizar)
+
+18. Criar workflow `wiki-sync.yml` espelhando `/docs` → `.wiki.git` (github-wiki-action) — **fim do drift manual**.
+19. Re-sincronizar a Wiki com os números corrigidos.
+
+### Fase G — CI de docs (fechar lacunas)
+
+20. **Adicionar count-guard** a `check:docs-counts`: providers, free, tests, locales, MCP tools/scopes → **gate blocking** (matando a causa-raiz).
+21. Promover `check-translation-drift` a blocking (`--strict`) após baseline.
+22. Adicionar job advisory `docs-lint` (Vale + markdownlint) e link-check externo (lychee, cron).
+23. Adicionar gate de consistência `config/i18n.json ↔ LANGUAGES`.
+
+---
+
+## 7. Decisões necessárias do usuário (antes de executar)
+
+1. **Headline de "free"**: manter `50+ / 11 forever` ou adotar número pesquisado (`98` documentados)?
+2. **Escopo dos novos docs**: criar todos P0+P1 agora, ou só P0 nesta rodada?
+3. **Wiki**: automatizar via workflow (recomendado) ou só re-sincronizar manualmente desta vez?
+4. **i18n**: re-traduzir os docs alterados agora (custa chamadas LLM) ou só corrigir o inglês e deixar i18n para um passo seguinte?
+5. **`in`/`pt` legados**: deprecar `in` (Indonésio legado) nesta rodada?
+6. **CI**: implementar os novos gates (count-guard, Vale, wiki-sync) nesta rodada ou em PR separado?
+
+---
+
+_Relatório gerado na Fase 1 (pesquisa). Nenhum documento de produto foi alterado ainda. A sincronização começa após confirmação do escopo acima._
diff --git a/docs/ops/meta.json b/docs/ops/meta.json
index b2248086f10..dfc1af0180f 100644
--- a/docs/ops/meta.json
+++ b/docs/ops/meta.json
@@ -8,6 +8,7 @@
"PROXY_GUIDE",
"SQLITE_RUNTIME",
"COVERAGE_PLAN",
- "E2E_DASHBOARD_SHAKEDOWN_v3.8.0"
+ "E2E_DASHBOARD_SHAKEDOWN_v3.8.0",
+ "DOCUMENTATION_AUDIT_REPORT"
]
}
diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md
index 7170e7c15fd..0929accccf0 100644
--- a/docs/reference/PROVIDER_REFERENCE.md
+++ b/docs/reference/PROVIDER_REFERENCE.md
@@ -1,18 +1,16 @@
---
title: "Provider Reference"
-version: 3.8.12
-lastUpdated: 2026-06-06
+version: 3.8.25
+lastUpdated: 2026-06-15
---
# Provider Reference
-> **For Users**: Looking for a simple guide? See the [Providers Guide](../getting-started/PROVIDERS-GUIDE.md) for step-by-step instructions.
-
> **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand.
> Regenerate with: `npm run gen:provider-reference`
-> **Last generated:** 2026-06-06
+> **Last generated:** 2026-06-15
-Total providers: **223**. See category breakdown below.
+Total providers: **226**. See category breakdown below.
## Categories
@@ -35,273 +33,276 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each
## OAuth Providers (19)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). |
-| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. |
-| `antigravity` | — | Antigravity | OAuth | — | — |
-| `claude` | `cc` | Claude Code | OAuth | — | — |
-| `cline` | `cl` | Cline | OAuth | — | — |
-| `codex` | `cx` | OpenAI Codex | OAuth | — | — |
-| `cursor` | `cu` | Cursor IDE | OAuth | — | — |
-| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai |
-| `gemini-cli` | `gemini-cli` | Gemini CLI | OAuth | — | Uses Gemini CLI OAuth / Cloud Code credentials. Pro models require an eligible Google account or paid plan. |
-| `github` | `gh` | GitHub Copilot | OAuth | — | — |
-| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | OAuth application with ai_features + read_user scopes. Configure GITLAB_DUO_OAUTH_CLIENT_ID and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET on this OmniRoute instance. |
-| `kilocode` | `kc` | Kilo Code | OAuth | — | — |
-| `kimi-coding` | `kmc` | Kimi Coding | OAuth | — | — |
-| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. |
-| `qoder` | `if` | Qoder AI | OAuth | — | — |
-| `qwen` | `qw` | Qwen Code | OAuth | — | ⚠️ **DEPRECATED.** Qwen OAuth free tier was discontinued on 2026-04-15. Use 'bailian-coding-plan', 'alibaba', 'alibaba-cn', or 'openrouter' provider with API key instead. |
-| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. |
-| `windsurf` | `ws` | Windsurf (Devin CLI) | OAuth | [link](https://windsurf.com) | Sign in at windsurf.com to get your token. Visit windsurf.com/show-auth-token after logging in and paste it here, or use the device-code login flow. |
-| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. |
+| ID | Alias | Name | Tags | Website | Notes |
+| ------------- | ------------ | -------------------- | ----- | ------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). |
+| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. |
+| `antigravity` | — | Antigravity | OAuth | — | — |
+| `claude` | `cc` | Claude Code | OAuth | — | — |
+| `cline` | `cl` | Cline | OAuth | — | — |
+| `codex` | `cx` | OpenAI Codex | OAuth | — | — |
+| `cursor` | `cu` | Cursor IDE | OAuth | — | — |
+| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai |
+| `gemini-cli` | `gemini-cli` | Gemini CLI | OAuth | — | Uses Gemini CLI OAuth / Cloud Code credentials. Pro models require an eligible Google account or paid plan. |
+| `github` | `gh` | GitHub Copilot | OAuth | — | — |
+| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | OAuth application with ai_features + read_user scopes. Configure GITLAB_DUO_OAUTH_CLIENT_ID and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET on this OmniRoute instance. |
+| `kilocode` | `kc` | Kilo Code | OAuth | — | — |
+| `kimi-coding` | `kmc` | Kimi Coding | OAuth | — | — |
+| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. |
+| `qoder` | `if` | Qoder AI | OAuth | — | — |
+| `qwen` | `qw` | Qwen Code | OAuth | — | ⚠️ **DEPRECATED.** Qwen OAuth free tier was discontinued on 2026-04-15. Use 'bailian-coding-plan', 'alibaba', 'alibaba-cn', or 'openrouter' provider with API key instead. |
+| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. |
+| `windsurf` | `ws` | Windsurf (Devin CLI) | OAuth | [link](https://windsurf.com) | In the Windsurf / VS Code IDE, open the command palette and run `Windsurf: Provide Auth Token` (or click the Jupyter "Get Windsurf Authentication Token" button), then copy the shown token and paste it here. Note: opening windsurf.com/show-auth-token directly only renders a "Redirecting" page — the IDE must initiate the flow (it adds a `?state=...` param) for the token to appear. |
+| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. |
-## Web Cookie Providers (20)
+## Web Cookie Providers (22)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) |
-| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai |
-| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com |
-| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai |
-| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste your access_token from copilot.microsoft.com (or export a .har file from DevTools while logged in) |
-| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken |
-| `doubao-web` | `db` | Doubao Web (ByteDance) | Web cookie | [link](https://www.doubao.com) | Paste your session cookie from doubao.com (DevTools → Application → Cookies) |
-| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. |
-| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. |
-| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste your hf-chat cookie value from huggingface.co/chat (DevTools → Application → Cookies → hf-chat). Optional — works without auth for basic use. |
-| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com |
-| `kimi-web` | `kimi-web` | Kimi Web (Moonshot AI) | Web cookie | [link](https://kimi.moonshot.cn) | Paste your session cookie from kimi.moonshot.cn (DevTools → Application → Cookies) |
-| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your abra_sess value or full cookie header from meta.ai |
-| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai |
-| `phind` | `ph` | Phind (Free) | Web cookie | [link](https://www.phind.com) | Paste your session cookie from phind.com (DevTools → Application → Cookies). Optional — works with free tier. |
-| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) |
-| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). |
-| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. |
-| `v0-vercel-web` | `v0` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) |
-| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) |
+| ID | Alias | Name | Tags | Website | Notes |
+| ----------------- | ------------- | ---------------------------- | ---------- | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your \_\_client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) |
+| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your \_\_Secure-authjs.session-token value or full cookie header from app.blackbox.ai |
+| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your \_\_Secure-next-auth.session-token cookie value from chatgpt.com |
+| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai |
+| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste your access_token from copilot.microsoft.com (or export a .har file from DevTools while logged in) |
+| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken |
+| `doubao-web` | `db` | Doubao Web (ByteDance) | Web cookie | [link](https://www.doubao.com) | Paste your session cookie from doubao.com (DevTools → Application → Cookies) |
+| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy **Secure-1PSID and **Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. |
+| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your **Secure-1PSID cookie value from gemini.google.com. Optionally add **Secure-1PSIDTS separated by semicolon. |
+| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. |
+| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste your hf-chat cookie value from huggingface.co/chat (DevTools → Application → Cookies → hf-chat). Optional — works without auth for basic use. |
+| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com |
+| `kimi-web` | `kimi-web` | Kimi Web (Moonshot AI) | Web cookie | [link](https://kimi.moonshot.cn) | Paste your session cookie from kimi.moonshot.cn (DevTools → Application → Cookies) |
+| `lmarena` | `lma` | LMArena (Free) | Web cookie | [link](https://lmarena.ai) | Paste your session cookie from lmarena.ai (DevTools → Application → Cookies). Optional — works with free tier for basic comparisons. |
+| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your abra_sess value or full cookie header from meta.ai |
+| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your \_\_Secure-next-auth.session-token cookie value from perplexity.ai |
+| `phind` | `ph` | Phind (Free) | Web cookie | [link](https://www.phind.com) | Paste your session cookie from phind.com (DevTools → Application → Cookies). Optional — works with free tier. |
+| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) |
+| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). |
+| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. |
+| `v0-vercel-web` | `v0` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) |
+| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) |
-## API Key Providers (paid / paid-with-free-credits) (151)
+## API Key Providers (paid / paid-with-free-credits) (152)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn |
-| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway |
-| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required |
-| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | $0.025/day free credits — 200+ models (GPT-4o, Claude, Gemini, Llama) via single endpoint |
-| `alibaba` | `ali` | Alibaba | API key | [link](https://dashscope-intl.aliyuncs.com) | — |
-| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.aliyuncs.com) | — |
-| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — |
-| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 |
-| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai |
-| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. |
-| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. |
-| `baichuan` | `baichuan` | Baichuan | API key | [link](https://baichuan.com) | Get API key at platform.baichuan-ai.com |
-| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://yiyan.baidu.com) | Get API key at console.bce.baidu.com |
-| `bailian-coding-plan` | `bcp` | Alibaba Coding Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/coding-plan) | — |
-| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference |
-| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Free tier with auto:free routing — zero-cost inference, no credit card required |
-| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. |
-| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — |
-| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Free tier: unlimited basic chat plus Minimax-M2.5, no credit card required |
-| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 |
-| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — |
-| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks |
-| `cablyai` | `cablyai` | CablyAI | API key, aggregator | [link](https://cablyai.com) | Bearer API key for the CablyAI OpenAI-compatible gateway. |
-| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. |
-| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. |
-| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . |
-| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) |
-| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — |
-| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required |
-| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. |
-| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api |
-| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — |
-| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — |
-| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. |
-| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration |
-| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required |
-| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. |
-| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com |
-| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. |
-| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — |
-| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required |
-| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. |
-| `firecrawl` | `fc` | Firecrawl | API key | [link](https://firecrawl.dev) | — |
-| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing |
-| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — |
-| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. |
-| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required |
-| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | — |
-| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com |
-| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — |
-| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — |
-| `github-models` | `ghm` | GitHub Models | API key | [link](https://github.com/marketplace/models) | Create a GitHub PAT with 'models: read' scope at github.com/settings/tokens |
-| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. |
-| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required |
-| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required |
-| `glhf` | `glhf` | GLHF Chat | API key, aggregator | [link](https://glhf.chat) | Bearer API key for the GLHF OpenAI-compatible gateway. |
-| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — |
-| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — |
-| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — |
-| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card |
-| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. |
-| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api |
-| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — |
-| `huggingchat` | `huggingchat` | HuggingChat | API key | [link](https://huggingface.co/chat) | No API key required for basic access. |
-| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) |
-| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference |
-| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api |
-| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn |
-| `inclusionai` | `inclusion` | InclusionAI | API key | [link](https://inclusionai.com) | Get API key at inclusionai.com |
-| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available |
-| `jina-ai` | `jina` | Jina AI | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for the Jina AI rerank API. |
-| `jina-reader` | `jr` | Jina Reader | API key | [link](https://jina.ai/reader) | — |
-| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — |
-| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — |
-| `kimi` | `kimi` | Kimi | API key | [link](https://platform.moonshot.ai) | — |
-| `kimi-coding-apikey` | `kmca` | Kimi Coding (API Key) | API key | [link](https://www.kimi.com/code) | — |
-| `kluster` | `kluster` | Kluster AI | API key | [link](https://kluster.ai) | $5 free credits on signup - DeepSeek R1, Llama 4 Maverick/Scout, Qwen3 235B |
-| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — |
-| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — |
-| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer |
-| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai |
-| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — |
-| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | No signup required - 2 req/s, 20 RPM, 100 req/hr free tier |
-| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: 5M tokens/day on LongCat-2.0-Preview (Flash models retired 2026-05-29); up to 120M/day via feedback. |
-| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — |
-| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — |
-| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — |
-| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — |
-| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required |
-| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. |
-| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | Get API key at monsterapi.ai |
-| `moonshot` | `moonshot` | Moonshot AI | API key | [link](https://platform.moonshot.ai) | — |
-| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 |
-| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — |
-| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing |
-| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. |
-| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai |
-| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. |
-| `novita` | `novita` | Novita AI | API key, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) |
-| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing |
-| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) |
-| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. |
-| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/api-keys) | — |
-| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — |
-| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — |
-| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — |
-| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD |
-| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — |
-| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — |
-| `phind` | `phind` | Phind | API key | [link](https://phind.com) | Get API key at phind.com |
-| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — |
-| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. |
-| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | No API key required for free public endpoint. Optional Spore tier: ~0.01 pollen/hour. |
-| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | $25 free trial credits (30-day validity) |
-| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Free community inference tier for testing |
-| `puter` | `pu` | Puter AI | API key | [link](https://puter.com) | Get token at puter.com/dashboard → Copy Auth Token |
-| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product/wenxinworkshop) | — |
-| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — |
-| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. |
-| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. |
-| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required |
-| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. |
-| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/ai/generative-apis) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B |
-| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn |
-| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus permanently free models after identity verification |
-| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — |
-| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn |
-| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — |
-| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com |
-| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) |
-| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — |
-| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com |
-| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. |
-| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | $25 signup credits + 3 permanently free models: Llama 3.3 70B, Vision, DeepSeek-R1 distill |
-| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — |
-| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) |
-| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. |
-| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — |
-| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — |
-| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — |
-| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — |
-| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token |
-| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. |
-| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — |
-| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. |
-| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — |
-| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. |
-| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | — |
-| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — |
-| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com |
-| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — |
+| ID | Alias | Name | Tags | Website | Notes |
+| --------------------- | -------------- | ------------------------------- | --------------------- | -------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn |
+| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway |
+| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required |
+| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | $0.025/day free credits — 200+ models (GPT-4o, Claude, Gemini, Llama) via single endpoint |
+| `alibaba` | `ali` | Alibaba | API key | [link](https://dashscope-intl.aliyuncs.com) | — |
+| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.aliyuncs.com) | — |
+| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — |
+| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 |
+| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai |
+| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. |
+| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. |
+| `baichuan` | `baichuan` | Baichuan | API key | [link](https://baichuan.com) | Get API key at platform.baichuan-ai.com |
+| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://yiyan.baidu.com) | Get API key at console.bce.baidu.com |
+| `bailian-coding-plan` | `bcp` | Alibaba Coding Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/coding-plan) | — |
+| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference |
+| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Free tier with auto:free routing — zero-cost inference, no credit card required |
+| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. |
+| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — |
+| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Free tier: unlimited basic chat plus Minimax-M2.5, no credit card required |
+| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 |
+| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — |
+| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks |
+| `cablyai` | `cablyai` | CablyAI | API key, aggregator | [link](https://cablyai.com) | Bearer API key for the CablyAI OpenAI-compatible gateway. |
+| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. |
+| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. |
+| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . |
+| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) |
+| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — |
+| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required |
+| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. |
+| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api |
+| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — |
+| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — |
+| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. |
+| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration |
+| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required |
+| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. |
+| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com |
+| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. |
+| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — |
+| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required |
+| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. |
+| `firecrawl` | `fc` | Firecrawl | API key | [link](https://firecrawl.dev) | — |
+| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing |
+| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — |
+| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. |
+| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required |
+| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | — |
+| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com |
+| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — |
+| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — |
+| `github-models` | `ghm` | GitHub Models | API key | [link](https://github.com/marketplace/models) | Create a GitHub PAT with 'models: read' scope at github.com/settings/tokens |
+| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. |
+| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required |
+| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required |
+| `glhf` | `glhf` | GLHF Chat | API key, aggregator | [link](https://glhf.chat) | Bearer API key for the GLHF OpenAI-compatible gateway. |
+| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — |
+| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — |
+| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — |
+| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card |
+| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. |
+| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api |
+| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — |
+| `huggingchat` | `huggingchat` | HuggingChat | API key | [link](https://huggingface.co/chat) | No API key required for basic access. |
+| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) |
+| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference |
+| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api |
+| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn |
+| `inclusionai` | `inclusion` | InclusionAI | API key | [link](https://inclusionai.com) | Get API key at inclusionai.com |
+| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available |
+| `jina-ai` | `jina` | Jina AI | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for the Jina AI rerank API. |
+| `jina-reader` | `jr` | Jina Reader | API key | [link](https://jina.ai/reader) | — |
+| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — |
+| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — |
+| `kimi` | `kimi` | Kimi | API key | [link](https://platform.moonshot.ai) | — |
+| `kimi-coding-apikey` | `kmca` | Kimi Coding (API Key) | API key | [link](https://www.kimi.com/code) | — |
+| `kluster` | `kluster` | Kluster AI | API key | [link](https://kluster.ai) | $5 free credits on signup - DeepSeek R1, Llama 4 Maverick/Scout, Qwen3 235B |
+| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — |
+| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — |
+| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer |
+| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai |
+| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — |
+| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | No signup required - 2 req/s, 20 RPM, 100 req/hr free tier |
+| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: 5M tokens/day on LongCat-2.0-Preview (Flash models retired 2026-05-29); up to 120M/day via feedback. |
+| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — |
+| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — |
+| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — |
+| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — |
+| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required |
+| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. |
+| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | Get API key at monsterapi.ai |
+| `moonshot` | `moonshot` | Moonshot AI | API key | [link](https://platform.moonshot.ai) | — |
+| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 |
+| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — |
+| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing |
+| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. |
+| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai |
+| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. |
+| `novita` | `novita` | Novita AI | API key, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) |
+| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing |
+| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) |
+| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. |
+| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/api-keys) | — |
+| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — |
+| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — |
+| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — |
+| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD |
+| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — |
+| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — |
+| `phind` | `phind` | Phind | API key | [link](https://phind.com) | Get API key at phind.com |
+| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — |
+| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. |
+| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | No API key required for free public endpoint. Optional Spore tier: ~0.01 pollen/hour. |
+| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | $25 free trial credits (30-day validity) |
+| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid |
+| `puter` | `pu` | Puter AI | API key | [link](https://puter.com) | Get token at puter.com/dashboard → Copy Auth Token |
+| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product/wenxinworkshop) | — |
+| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — |
+| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. |
+| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. |
+| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required |
+| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. |
+| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/ai/generative-apis) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B |
+| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn |
+| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus permanently free models after identity verification |
+| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — |
+| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn |
+| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — |
+| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com |
+| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) |
+| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — |
+| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com |
+| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. |
+| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | $25 signup credits + 3 permanently free models: Llama 3.3 70B, Vision, DeepSeek-R1 distill |
+| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — |
+| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) |
+| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. |
+| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — |
+| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — |
+| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — |
+| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — |
+| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token |
+| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. |
+| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — |
+| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. |
+| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — |
+| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. |
+| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | — |
+| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — |
+| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com |
+| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — |
+| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. |
## Local Providers (11)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). |
-| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). |
-| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). |
-| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. |
-| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). |
-| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). |
-| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). |
-| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). |
-| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). |
-| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). |
-| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). |
+| ID | Alias | Name | Tags | Website | Notes |
+| --------------------- | ------------ | ------------------- | ------------------ | --------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). |
+| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). |
+| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). |
+| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. |
+| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). |
+| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). |
+| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). |
+| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). |
+| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). |
+| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). |
+| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). |
## Search Providers (11)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard |
-| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai |
-| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) |
-| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard |
-| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/api-keys) | Same API key as Ollama Cloud (from ollama.com/settings/api-keys) |
-| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) |
-| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs) | API key from SearchAPI (query param or Bearer auth) |
-| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. |
-| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard |
-| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) |
-| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/docs/search/overview) | X-API-Key from the You.com platform dashboard |
+| ID | Alias | Name | Tags | Website | Notes |
+| ------------------- | --------------- | -------------------------- | ------ | --------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- |
+| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard |
+| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai |
+| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) |
+| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard |
+| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/api-keys) | Same API key as Ollama Cloud (from ollama.com/settings/api-keys) |
+| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) |
+| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs) | API key from SearchAPI (query param or Bearer auth) |
+| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. |
+| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard |
+| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) |
+| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/docs/search/overview) | X-API-Key from the You.com platform dashboard |
## Audio-only Providers (7)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — |
-| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. |
-| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — |
-| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — |
-| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — |
-| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — |
-| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — |
+| ID | Alias | Name | Tags | Website | Notes |
+| ------------ | ---------- | ---------- | ----- | ------------------------------------- | ----------------------------------------------------------------------------------------------- |
+| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — |
+| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. |
+| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — |
+| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — |
+| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — |
+| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — |
+| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — |
## Upstream Proxy Providers (2)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — |
-| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — |
+| ID | Alias | Name | Tags | Website | Notes |
+| ------------- | ----- | ----------- | -------------- | ---------------------------------------------------- | ----- |
+| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — |
+| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — |
## Cloud Agent Providers (3)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. |
-| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. |
-| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. |
+| ID | Alias | Name | Tags | Website | Notes |
+| ------------- | ------------- | ------------ | ----------- | -------------------------------- | ----------------------------------------------------------- |
+| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. |
+| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. |
+| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. |
## System Providers (1)
-| ID | Alias | Name | Tags | Website | Notes |
-|----|-------|------|------|---------|-------|
-| `auto` | `auto` | Auto (Zero-Config) | System | — | — |
+| ID | Alias | Name | Tags | Website | Notes |
+| ------ | ------ | ------------------ | ------ | ------- | ----- |
+| `auto` | `auto` | Auto (Zero-Config) | System | — | — |
## Sources of truth
diff --git a/docs/reference/openapi.yaml b/docs/reference/openapi.yaml
index 2a4c081486d..102fef12b00 100644
--- a/docs/reference/openapi.yaml
+++ b/docs/reference/openapi.yaml
@@ -1,7 +1,7 @@
openapi: 3.1.0
info:
title: OmniRoute API
- version: 3.8.25
+ version: 3.8.26
description: |
OmniRoute is a local-first AI API proxy router. It provides an OpenAI-compatible
endpoint that routes requests to multiple AI providers with load balancing,
diff --git a/electron/package-lock.json b/electron/package-lock.json
index 4adfd09b33e..9a2e3e1b74e 100644
--- a/electron/package-lock.json
+++ b/electron/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "omniroute-desktop",
- "version": "3.8.25",
+ "version": "3.8.26",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "omniroute-desktop",
- "version": "3.8.25",
+ "version": "3.8.26",
"license": "MIT",
"dependencies": {
"electron-updater": "^6.8.9"
diff --git a/electron/package.json b/electron/package.json
index a4346545921..9a4b6b138b6 100644
--- a/electron/package.json
+++ b/electron/package.json
@@ -1,6 +1,6 @@
{
"name": "omniroute-desktop",
- "version": "3.8.25",
+ "version": "3.8.26",
"description": "OmniRoute Desktop Application",
"main": "main.js",
"author": {
diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts
index 26b0e833a8a..e18b5faa47c 100644
--- a/open-sse/config/glmProvider.ts
+++ b/open-sse/config/glmProvider.ts
@@ -16,6 +16,30 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({
});
export const GLM_SHARED_MODELS = Object.freeze([
+ {
+ id: "glm-5.2",
+ name: "GLM 5.2",
+ contextLength: 1000000,
+ maxOutputTokens: 131072,
+ toolCalling: true,
+ supportsReasoning: true,
+ },
+ {
+ id: "glm-5.2-high",
+ name: "GLM 5.2 High",
+ contextLength: 1000000,
+ maxOutputTokens: 131072,
+ toolCalling: true,
+ supportsReasoning: true,
+ },
+ {
+ id: "glm-5.2-max",
+ name: "GLM 5.2 Max",
+ contextLength: 1000000,
+ maxOutputTokens: 131072,
+ toolCalling: true,
+ supportsReasoning: true,
+ },
{
id: "glm-5.1",
name: "GLM 5.1",
diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts
index 5dfa8495d90..07311134d3c 100644
--- a/open-sse/config/providerRegistry.ts
+++ b/open-sse/config/providerRegistry.ts
@@ -4535,6 +4535,11 @@ export function generateModels(): Record {
if (!models[key]) {
models[key] = entry.models;
}
+ // Also store under the raw provider id so getProviderModels(id) works
+ // even when the provider has a different alias (e.g. "github" → alias "gh").
+ if (entry.alias && entry.alias !== entry.id && !models[entry.id]) {
+ models[entry.id] = entry.models;
+ }
}
}
return models;
diff --git a/open-sse/executors/default.ts b/open-sse/executors/default.ts
index 2ec6fbd6b3d..a585c0a72f2 100644
--- a/open-sse/executors/default.ts
+++ b/open-sse/executors/default.ts
@@ -151,6 +151,14 @@ function normalizeOpenAIChatUrl(baseUrl) {
return `${normalized}/v1/chat/completions`;
}
+function getOpenRouterConnectionPreset(
+ providerSpecificData?: Record | null
+): string | null {
+ const preset =
+ typeof providerSpecificData?.preset === "string" ? providerSpecificData.preset.trim() : "";
+ return preset || null;
+}
+
export class DefaultExecutor extends BaseExecutor {
constructor(provider) {
super(provider, PROVIDERS[provider] || PROVIDERS.openai);
@@ -564,6 +572,16 @@ export class DefaultExecutor extends BaseExecutor {
}
}
}
+
+ if (this.provider === "openrouter") {
+ const connectionPreset = getOpenRouterConnectionPreset(credentials?.providerSpecificData);
+ if (connectionPreset && (withDefaults as Record).preset === undefined) {
+ withDefaults = {
+ ...(withDefaults as Record),
+ preset: connectionPreset,
+ };
+ }
+ }
}
if (this.provider === "qwen" && typeof withDefaults === "object" && withDefaults !== null) {
diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts
index 6e78ead8db8..f83c6ae260c 100644
--- a/open-sse/executors/glm.ts
+++ b/open-sse/executors/glm.ts
@@ -50,6 +50,19 @@ function getEffectiveKey(credentials: ProviderCredentials): string {
return credentials.apiKey || credentials.accessToken || "";
}
+/**
+ * GLM-5.2 effort tiers route exclusively through the Anthropic transport,
+ * where Zhipu maps Claude Code effort selectors (high/max) to reasoning
+ * intensity. The base model ID sent upstream is always "glm-5.2".
+ *
+ * https://docs.z.ai/devpack/latest-model
+ */
+function parseGlm52Effort(model: string): { baseModel: string; effort: "high" | "max" } | null {
+ if (model === "glm-5.2-high") return { baseModel: "glm-5.2", effort: "high" };
+ if (model === "glm-5.2-max") return { baseModel: "glm-5.2", effort: "max" };
+ return null;
+}
+
function applyGlmRequestDefaults(body: unknown, defaults?: JsonRecord | null): unknown {
const record = asRecord(body);
if (!record || !defaults) return body;
@@ -228,27 +241,61 @@ export class GlmExecutor extends DefaultExecutor {
credentials: ProviderCredentials,
transport: GlmTransport
) {
- const transformed = this.transformRequest(model, body, stream, credentials);
+ const effortTier = parseGlm52Effort(model);
+ const effectiveModel = effortTier ? effortTier.baseModel : model;
+
+ const transformed = this.transformRequest(effectiveModel, body, stream, credentials);
+ const record = asRecord(transformed);
+
+ // Ensure upstream receives the base model ID, not the effort-suffixed alias
+ if (record && effortTier) {
+ record.model = effectiveModel;
+ }
if (transport === "openai") {
- const record = asRecord(transformed);
if (record && stream && hasTools(record) && record.tool_stream === undefined) {
return { ...record, tool_stream: true };
}
return transformed;
}
- return translateRequest(
+ const translated = translateRequest(
FORMATS.OPENAI,
FORMATS.CLAUDE,
- model,
- { ...(transformed as JsonRecord), _disableToolPrefix: true },
+ effectiveModel,
+ { ...(record ?? {}), _disableToolPrefix: true },
stream,
credentials,
this.provider,
null,
{ preserveCacheControl: false }
);
+
+ // Inject effort and thinking for the Anthropic transport.
+ // Zhipu's Anthropic endpoint requires thinking.type=enabled to emit
+ // thinking_delta blocks in the SSE response. Without it, reasoning is
+ // not surfaced and clients see no thinking content.
+ // The effort-2025-11-24 beta header (in GLM_ANTHROPIC_BETA) carries
+ // the high/max intensity selector.
+ if (effortTier) {
+ const translatedRecord = asRecord(translated);
+ if (translatedRecord) {
+ translatedRecord.effort = effortTier.effort;
+ // Zhipu's Anthropic endpoint only supports thinking.type
+ // "enabled"/"disabled" — not "adaptive". Clients like Claude Code
+ // default to "adaptive" for reasoning models, so force "enabled"
+ // here while preserving any other fields (e.g. budget_tokens).
+ const existingThinking = asRecord(translatedRecord.thinking);
+ if (!existingThinking || existingThinking.type !== "enabled") {
+ translatedRecord.thinking = {
+ ...existingThinking,
+ type: "enabled",
+ };
+ }
+ }
+ }
+
+ return translated;
}
private async executeTransport(
@@ -343,6 +390,15 @@ export class GlmExecutor extends DefaultExecutor {
}
async execute(input: ExecuteInput): Promise {
+ const effortTier = parseGlm52Effort(input.model);
+
+ // GLM-5.2 effort tiers route directly through Anthropic transport (no fallback).
+ // Zhipu only graduates effort on the Anthropic endpoint via the
+ // effort-2025-11-24 beta header included in GLM_ANTHROPIC_BETA.
+ if (effortTier) {
+ return this.executeTransport(input, "anthropic");
+ }
+
const primaryTransport = getGlmTransport(
input.credentials.providerSpecificData,
this.config.baseUrl
diff --git a/open-sse/mcp-server/__tests__/audit.test.ts b/open-sse/mcp-server/__tests__/audit.test.ts
index 90e5713a589..d51541f17f4 100644
--- a/open-sse/mcp-server/__tests__/audit.test.ts
+++ b/open-sse/mcp-server/__tests__/audit.test.ts
@@ -86,4 +86,57 @@ describe("MCP audit shutdown", () => {
expect(audit.closeAuditDb()).toBe(true);
expect(mockDb.close).toHaveBeenCalledTimes(1);
});
+
+ it("falls back to node:sqlite when better-sqlite3 binding is missing", async () => {
+ const [maj, min] = process.versions.node.split(".").map(Number);
+ if (maj < 22 || (maj === 22 && min < 5)) {
+ return; // node:sqlite not available on this Node, skip
+ }
+
+ // Simulate a global-install scenario where the bundled native binary
+ // never landed in dist/node_modules/better-sqlite3/build/Release/.
+ const bindingErr = new Error(
+ "Could not locate the bindings file. Tried: …/better_sqlite3.node"
+ ) as Error & { code?: string };
+ bindingErr.code = "MODULE_NOT_FOUND";
+ // Simulate the binding-missing failure as the better-sqlite3 default
+ // constructor throwing — this matches reality (`new Database()` throws
+ // "Could not locate the bindings file" when the prebuilt .node is absent)
+ // and reaches the adapter's `catch (nativeErr)`. A factory that itself
+ // throws is reported by vitest as a mock-setup error and never reaches
+ // the code under test.
+ const ThrowingDatabase = vi.fn(function ThrowingDatabase() {
+ throw bindingErr;
+ });
+ vi.doMock("better-sqlite3", () => ({
+ default: ThrowingDatabase,
+ }));
+
+ // node:sqlite's DatabaseSync does not expose a boolean `open` property,
+ // so the mock intentionally omits it — the adapter tracks open state in
+ // a local closure and exposes it via a getter.
+ const mockNodeDb = {
+ prepare: vi.fn(() => createStatementMock()),
+ exec: vi.fn(),
+ close: vi.fn(),
+ };
+ const DatabaseSync = vi.fn(function DatabaseSync() {
+ return mockNodeDb;
+ });
+ vi.doMock("node:sqlite", () => ({ DatabaseSync }));
+
+ const audit = await import("../audit.ts");
+
+ await audit.logToolCall("omniroute_get_health", { ok: true }, { ok: true }, 4, true);
+ expect(DatabaseSync).toHaveBeenCalledWith(dbFile);
+ expect(mockNodeDb.prepare).toHaveBeenCalled();
+
+ expect(audit.closeAuditDb()).toBe(true);
+ expect(mockNodeDb.exec).toHaveBeenCalledWith("PRAGMA wal_checkpoint(TRUNCATE)");
+ expect(mockNodeDb.close).toHaveBeenCalledTimes(1);
+
+ // Cache is cleared after close, so a second close is a no-op.
+ expect(audit.closeAuditDb()).toBe(false);
+ expect(mockNodeDb.close).toHaveBeenCalledTimes(1);
+ });
});
diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts
index f937df26c24..7a2db82b6da 100644
--- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts
+++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts
@@ -88,6 +88,9 @@ describe("GLM Coding provider registry surfaces", () => {
expect(PROVIDER_ID_TO_ALIAS.glm).toBe("glm");
expect(byProviderId).toEqual(byAlias);
expect(byProviderId.map((model) => model.id)).toEqual([
+ "glm-5.2",
+ "glm-5.2-high",
+ "glm-5.2-max",
"glm-5.1",
"glm-5",
"glm-5-turbo",
@@ -101,6 +104,30 @@ describe("GLM Coding provider registry surfaces", () => {
]);
});
+ it("registers GLM-5.2 with correct specs and effort tier aliases", () => {
+ const models = getModelsByProviderId("glm");
+ const get = (id: string) => models.find((m) => m.id === id);
+
+ // Base model
+ const base = get("glm-5.2");
+ expect(base).toBeDefined();
+ expect(base?.contextLength).toBe(1000000);
+ expect(base?.maxOutputTokens).toBe(131072);
+ expect(base?.supportsReasoning).toBe(true);
+ expect(base?.toolCalling).toBe(true);
+
+ // Effort tier aliases share the same specs
+ const high = get("glm-5.2-high");
+ expect(high).toBeDefined();
+ expect(high?.contextLength).toBe(1000000);
+ expect(high?.maxOutputTokens).toBe(131072);
+
+ const max = get("glm-5.2-max");
+ expect(max).toBeDefined();
+ expect(max?.contextLength).toBe(1000000);
+ expect(max?.maxOutputTokens).toBe(131072);
+ });
+
it("applies doc-backed context window overrides for GLM models", () => {
const models = getModelsByProviderId("glm");
const get = (id: string) => models.find((m) => m.id === id);
@@ -126,6 +153,9 @@ describe("GLM Coding provider registry surfaces", () => {
expect(supportsToolCalling("glm/glm-5")).toBe(true);
expect(supportsToolCalling("glm/glm-4.7-flash")).toBe(true);
expect(supportsToolCalling("glm/glm-4.5-air")).toBe(true);
+ expect(supportsToolCalling("glm/glm-5.2")).toBe(true);
+ expect(supportsToolCalling("glm/glm-5.2-high")).toBe(true);
+ expect(supportsToolCalling("glm/glm-5.2-max")).toBe(true);
expect(getPricingForModel("glm", "glm-5")).toEqual({
input: 1.0,
@@ -148,6 +178,20 @@ describe("GLM Coding provider registry surfaces", () => {
reasoning: 1.1,
cache_creation: 0.2,
});
+ expect(getPricingForModel("glm", "glm-5.2")).toEqual({
+ input: 1.2,
+ output: 5,
+ cached: 0.3,
+ reasoning: 5,
+ cache_creation: 1.2,
+ });
+ expect(getPricingForModel("glm", "glm-5.2-max")).toEqual({
+ input: 1.2,
+ output: 5,
+ cached: 0.3,
+ reasoning: 5,
+ cache_creation: 1.2,
+ });
});
it("keeps the repo-derived GLM inventory internally aligned across registry and pricing surfaces", () => {
diff --git a/open-sse/mcp-server/audit.ts b/open-sse/mcp-server/audit.ts
index 18897035e06..259ded1c49f 100644
--- a/open-sse/mcp-server/audit.ts
+++ b/open-sse/mcp-server/audit.ts
@@ -7,6 +7,7 @@
*/
import { hashInput, summarizeOutput } from "./schemas/audit.ts";
+import { isNativeSqliteLoadError } from "../../src/lib/db/core.ts";
// ============ Database Connection ============
@@ -21,6 +22,61 @@ interface AuditDatabase {
pragma: (sql: string) => unknown;
close: () => void;
open?: boolean;
+ driver?: "better-sqlite3" | "node:sqlite";
+}
+
+interface NodeSqliteDatabase {
+ prepare: (sql: string) => {
+ run: (...params: unknown[]) => { changes: number | bigint; lastInsertRowid: number | bigint };
+ get: (...params: unknown[]) => unknown;
+ all: (...params: unknown[]) => unknown[];
+ };
+ exec: (sql: string) => void;
+ close: () => void;
+}
+
+/**
+ * node:sqlite's `DatabaseSync` does NOT expose a boolean `open` property —
+ * `open` and `close` are methods on the prototype, and the only state
+ * surface is the `isOpen` getter. Track open state locally in a closure
+ * so the adapter's `AuditDatabase` contract (`open?: boolean`) is honored
+ * and `getCachedAuditDb()`'s truthy check doesn't return a closed handle
+ * after `closeAuditDb()`.
+ */
+function createNodeSqliteAuditAdapter(db: NodeSqliteDatabase): AuditDatabase {
+ let _isOpen = true;
+ return {
+ driver: "node:sqlite",
+ get open() {
+ return _isOpen;
+ },
+ prepare(sql: string) {
+ const stmt = db.prepare(sql);
+ return {
+ get: (...params: unknown[]) => stmt.get(...params) as TRow | undefined,
+ all: (...params: unknown[]) => stmt.all(...params) as TRow[],
+ run: (...params: unknown[]) => stmt.run(...params),
+ };
+ },
+ pragma(pragmaSql: string) {
+ // node:sqlite has no .pragma() helper — route through .exec() for
+ // statement-shaped PRAGMAs (e.g. "wal_checkpoint(TRUNCATE)").
+ try {
+ db.exec(`PRAGMA ${pragmaSql}`);
+ return null;
+ } catch (err) {
+ return err instanceof Error ? err.message : String(err);
+ }
+ },
+ close: () => {
+ if (!_isOpen) return;
+ try {
+ db.close();
+ } finally {
+ _isOpen = false;
+ }
+ },
+ };
}
declare global {
@@ -153,6 +209,15 @@ function toString(value: unknown): string {
/**
* Lazy-load the database connection.
* Uses the same SQLite database as the main OmniRoute app.
+ *
+ * Driver priority:
+ * 1. better-sqlite3 — fast native binding (when its compiled `.node`
+ * binary is present, see scripts/build/postinstall.mjs).
+ * 2. node:sqlite — built-in to Node 22.5+. Used as a transparent
+ * fallback so the MCP audit logger still works on installs where
+ * the better-sqlite3 binary failed to resolve (e.g. missing
+ * `dist/node_modules/better-sqlite3/build/Release/better_sqlite3.node`
+ * in some global-install / Docker scenarios).
*/
async function getDb(): Promise {
const cachedDb = getCachedAuditDb();
@@ -173,12 +238,59 @@ async function getDb(): Promise {
return null;
}
- const Database = (await import("better-sqlite3")).default as unknown as new (
- dbPath: string
- ) => AuditDatabase;
- const database = new Database(dbPath);
- setCachedAuditDb(database);
- return database;
+ // Try better-sqlite3 first (matches the main app's default driver).
+ try {
+ const Database = (await import("better-sqlite3")).default as unknown as new (
+ dbPath: string
+ ) => AuditDatabase;
+ const database = new Database(dbPath);
+ setCachedAuditDb(database);
+ return database;
+ } catch (nativeErr) {
+ // Declared once at the top of the catch: nativeMessage is read both on
+ // the non-fallback bail-out and in the node:sqlite fallback warning
+ // further down. A block-scoped const inside the `if` below would be out
+ // of scope in the fallback path.
+ const nativeMessage = nativeErr instanceof Error ? nativeErr.message : String(nativeErr);
+ // Reuse the canonical detection helper from the main app's DB layer
+ // so we cover every ABI/binding failure mode the rest of the codebase
+ // already knows about: missing MODULE_NOT_FOUND, ERR_DLOPEN_FAILED,
+ // "Module did not self-register", "Cannot find module 'better-sqlite3'",
+ // the standard V8 "was compiled against a different Node.js version"
+ // message, and the bindings-loader "Could not locate the bindings file".
+ // Real errors (corrupt db, permission denied) still surface to the operator.
+ if (!isNativeSqliteLoadError(nativeErr)) {
+ console.error("[MCP Audit] Failed to connect to database:", nativeMessage);
+ return null;
+ }
+ // Fall back to Node's built-in sqlite (Node 22.5+).
+ const [maj, min] = (process.versions.node ?? "0.0").split(".").map(Number);
+ if (maj < 22 || (maj === 22 && (min ?? 0) < 5)) {
+ console.error(
+ `[MCP Audit] better-sqlite3 native binding unavailable and Node ${process.version} ` +
+ "has no built-in sqlite. Audit logging disabled. Fix: run " +
+ "`npm rebuild better-sqlite3` in the omniroute install root."
+ );
+ return null;
+ }
+ try {
+ const { DatabaseSync } = (await import("node:sqlite")) as {
+ DatabaseSync: new (p: string) => NodeSqliteDatabase;
+ };
+ const nodeDb = new DatabaseSync(dbPath);
+ const adapter = createNodeSqliteAuditAdapter(nodeDb);
+ setCachedAuditDb(adapter);
+ console.warn(
+ `[MCP Audit] better-sqlite3 binding unavailable — fell back to node:sqlite ` +
+ `(${nativeMessage.split("\n")[0]})`
+ );
+ return adapter;
+ } catch (nodeErr) {
+ const nodeMessage = nodeErr instanceof Error ? nodeErr.message : String(nodeErr);
+ console.error("[MCP Audit] Failed to connect to database:", nodeMessage);
+ return null;
+ }
+ }
} catch (err: unknown) {
const message = err instanceof Error ? err.message : String(err);
console.error("[MCP Audit] Failed to connect to database:", message);
diff --git a/open-sse/package.json b/open-sse/package.json
index 10ad729f0da..22fc9005dee 100644
--- a/open-sse/package.json
+++ b/open-sse/package.json
@@ -1,6 +1,6 @@
{
"name": "@omniroute/open-sse",
- "version": "3.8.25",
+ "version": "3.8.26",
"description": "Express SSE sidecar for OmniRoute — handles streaming, protocol translation, and provider orchestration",
"type": "module",
"main": "index.js",
diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts
index 402f0330967..7d5be3aca87 100644
--- a/open-sse/services/combo.ts
+++ b/open-sse/services/combo.ts
@@ -828,6 +828,7 @@ const MAX_RR_COUNTERS = 500;
const MAX_RESET_AWARE_CACHE = 200;
const rrCounters = new Map();
+const rrStickyTargets = new Map();
const resetAwareConnectionCache = new Map<
string,
@@ -849,6 +850,50 @@ function normalizeModelEntry(entry: unknown): { model: string; weight: number }
};
}
+function clampStickyRoundRobinTargetLimit(value: unknown): number {
+ const numericValue = Number(value);
+ if (!Number.isFinite(numericValue)) return 1;
+ return Math.min(Math.max(Math.floor(numericValue), 1), 1000);
+}
+
+function getStickyRoundRobinStartIndex(
+ comboName: string,
+ targets: ResolvedComboTarget[],
+ stickyLimit: number
+): { startIndex: number; counter: number } {
+ const sticky = rrStickyTargets.get(comboName);
+ const stickyIndex = sticky
+ ? targets.findIndex((target) => target.executionKey === sticky.executionKey)
+ : -1;
+ if (stickyLimit > 1 && sticky && stickyIndex >= 0 && sticky.successCount < stickyLimit) {
+ return { startIndex: stickyIndex, counter: rrCounters.get(comboName) || 0 };
+ }
+
+ const counter = rrCounters.get(comboName) || 0;
+ return { startIndex: counter % targets.length, counter };
+}
+
+function recordStickyRoundRobinSuccess(
+ comboName: string,
+ target: ResolvedComboTarget,
+ stickyLimit: number,
+ targets: ResolvedComboTarget[]
+): void {
+ const sticky = rrStickyTargets.get(comboName);
+ const successCount = sticky?.executionKey === target.executionKey ? sticky.successCount + 1 : 1;
+ if (successCount >= stickyLimit) {
+ const servedIndex = targets.findIndex((entry) => entry.executionKey === target.executionKey);
+ rrCounters.set(
+ comboName,
+ servedIndex >= 0 ? servedIndex + 1 : (rrCounters.get(comboName) || 0) + 1
+ );
+ rrStickyTargets.delete(comboName);
+ return;
+ }
+
+ rrStickyTargets.set(comboName, { executionKey: target.executionKey, successCount });
+}
+
function getTargetProvider(modelStr: string, providerId?: string | null): string {
const parsed = parseModel(modelStr);
return providerId || parsed.provider || parsed.providerAlias || "unknown";
@@ -2994,7 +3039,8 @@ export async function expandAutoComboCandidatePool(
(combo?.config as Record | undefined) ||
{};
- if (Array.isArray(localAutoConfig?.candidatePool)) return eligibleTargets;
+ if (Array.isArray(localAutoConfig?.candidatePool) && localAutoConfig.candidatePool.length > 0)
+ return eligibleTargets;
try {
const allConnections = await getProviderConnections({ isActive: true });
@@ -4700,14 +4746,38 @@ async function handleRoundRobinCombo({
log
);
- // Get and increment atomic counter
- const counter = rrCounters.get(combo.name) || 0;
- if (!rrCounters.has(combo.name) && rrCounters.size >= MAX_RR_COUNTERS) {
+ // Sticky batch size at the combo level. Reuses the global `stickyRoundRobinLimit`
+ // setting so a single knob controls sticky batching for both account fallback and
+ // combo targets. Values <= 1 preserve the historical one-request-per-target rotation.
+ const stickyLimit = clampStickyRoundRobinTargetLimit(
+ (settings as Record | null)?.stickyRoundRobinLimit
+ );
+ const stickyRoundRobinEnabled = stickyLimit > 1;
+ if (
+ !rrCounters.has(combo.name) &&
+ !rrStickyTargets.has(combo.name) &&
+ rrCounters.size >= MAX_RR_COUNTERS
+ ) {
const oldest = rrCounters.keys().next().value;
- if (oldest !== undefined) rrCounters.delete(oldest);
+ if (oldest !== undefined) {
+ rrCounters.delete(oldest);
+ rrStickyTargets.delete(oldest);
+ }
+ }
+ // Ensure rrCounters has an entry for this combo so the eviction logic above
+ // applies to both maps even when sticky round-robin is enabled (in which
+ // case rrCounters isn't incremented per request).
+ if (!rrCounters.has(combo.name)) {
+ rrCounters.set(combo.name, 0);
+ }
+ const { startIndex, counter } = getStickyRoundRobinStartIndex(
+ combo.name,
+ filteredTargets,
+ stickyLimit
+ );
+ if (!stickyRoundRobinEnabled) {
+ rrCounters.set(combo.name, counter + 1);
}
- rrCounters.set(combo.name, counter + 1);
- const startIndex = counter % modelCount;
const clientRequestedStream = body?.stream === true;
const startTime = Date.now();
@@ -4903,6 +4973,10 @@ async function handleRoundRobinCombo({
recordProviderSuccess(provider, target.connectionId ?? undefined);
}
+ if (stickyRoundRobinEnabled) {
+ recordStickyRoundRobinSuccess(combo.name, target, stickyLimit, filteredTargets);
+ }
+
if (provider) {
const connId = target.connectionId || undefined;
void (async () => {
diff --git a/open-sse/utils/stream/claudeLifecycle.ts b/open-sse/utils/stream/claudeLifecycle.ts
new file mode 100644
index 00000000000..3c2b34bb992
--- /dev/null
+++ b/open-sse/utils/stream/claudeLifecycle.ts
@@ -0,0 +1,180 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { JsonRecord } from "./types.ts";
+
+export type ClaudeEmptyResponseLifecycle = {
+ hasMessageStart: boolean;
+ hasContentBlock: boolean;
+ hasMessageDelta: boolean;
+ hasMessageStop: boolean;
+ hasError: boolean;
+ syntheticContentInjected: boolean;
+ warningLogged: boolean;
+};
+
+export const SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT = "";
+
+export function createClaudeEmptyResponseLifecycle(): ClaudeEmptyResponseLifecycle {
+ return {
+ hasMessageStart: false,
+ hasContentBlock: false,
+ hasMessageDelta: false,
+ hasMessageStop: false,
+ hasError: false,
+ syntheticContentInjected: false,
+ warningLogged: false,
+ };
+}
+
+export function getClaudeEventType(payload: unknown): string | null {
+ if (!payload || typeof payload !== "object") return null;
+ const type = (payload as JsonRecord).type;
+ return typeof type === "string" ? type : null;
+}
+
+export function isClaudeEventPayload(payload: unknown): payload is JsonRecord {
+ return getClaudeEventType(payload) !== null;
+}
+
+export function updateClaudeEmptyResponseLifecycle(
+ lifecycle: ClaudeEmptyResponseLifecycle,
+ payload: unknown
+) {
+ const type = getClaudeEventType(payload);
+ if (!type) return;
+
+ switch (type) {
+ case "message_start":
+ lifecycle.hasMessageStart = true;
+ break;
+ case "content_block_start":
+ case "content_block_delta":
+ case "content_block_stop":
+ lifecycle.hasContentBlock = true;
+ break;
+ case "message_delta":
+ lifecycle.hasMessageDelta = true;
+ break;
+ case "message_stop":
+ lifecycle.hasMessageStop = true;
+ break;
+ case "error":
+ lifecycle.hasError = true;
+ break;
+ default:
+ break;
+ }
+}
+
+export function hasClaudeAssistantLifecycle(lifecycle: ClaudeEmptyResponseLifecycle): boolean {
+ return lifecycle.hasMessageStart || lifecycle.hasMessageDelta || lifecycle.hasMessageStop;
+}
+
+export function shouldInjectClaudeEmptyResponseBeforeCurrentEvent(
+ lifecycle: ClaudeEmptyResponseLifecycle,
+ payload: unknown
+): boolean {
+ const type = getClaudeEventType(payload);
+ if (!type || lifecycle.hasError || lifecycle.hasContentBlock) return false;
+ if (!hasClaudeAssistantLifecycle(lifecycle)) return false;
+ return type === "message_delta" || type === "message_stop";
+}
+
+export function shouldInjectClaudeEmptyResponseOnFlush(lifecycle: ClaudeEmptyResponseLifecycle): boolean {
+ if (lifecycle.hasError || lifecycle.hasContentBlock) return false;
+ return hasClaudeAssistantLifecycle(lifecycle);
+}
+
+export function shouldInjectClaudeMissingFinalizersOnFlush(
+ lifecycle: ClaudeEmptyResponseLifecycle
+): boolean {
+ if (lifecycle.hasError || !lifecycle.syntheticContentInjected) return false;
+ return !lifecycle.hasMessageDelta || !lifecycle.hasMessageStop;
+}
+
+export function buildSyntheticClaudeEmptyResponseEvents(
+ lifecycle: ClaudeEmptyResponseLifecycle,
+ model: string | null,
+ options: {
+ includeContentBlock?: boolean;
+ includeMessageDelta?: boolean;
+ includeMessageStop?: boolean;
+ } = {}
+): JsonRecord[] {
+ const {
+ includeContentBlock = true,
+ includeMessageDelta = false,
+ includeMessageStop = false,
+ } = options;
+ const events: JsonRecord[] = [];
+ const resolvedModel = typeof model === "string" && model ? model : "unknown";
+
+ if (includeContentBlock) {
+ if (!lifecycle.hasMessageStart) {
+ events.push({
+ type: "message_start",
+ message: {
+ id: `msg_synthetic_${Date.now()}`,
+ type: "message",
+ role: "assistant",
+ model: resolvedModel,
+ content: [],
+ stop_reason: null,
+ stop_sequence: null,
+ usage: { input_tokens: 0, output_tokens: 0 },
+ },
+ });
+ }
+
+ events.push(
+ {
+ type: "content_block_start",
+ index: 0,
+ content_block: { type: "text", text: "" },
+ },
+ {
+ type: "content_block_delta",
+ index: 0,
+ delta: {
+ type: "text_delta",
+ text: SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT,
+ },
+ },
+ {
+ type: "content_block_stop",
+ index: 0,
+ }
+ );
+ }
+
+ if (includeMessageDelta) {
+ events.push({
+ type: "message_delta",
+ delta: { stop_reason: "end_turn", stop_sequence: null },
+ usage: { input_tokens: 0, output_tokens: 0 },
+ });
+ }
+
+ if (includeMessageStop) {
+ events.push({ type: "message_stop" });
+ }
+
+ return events;
+}
+
+export function restoreClaudePassthroughToolUseName(parsed: JsonRecord, toolNameMap: unknown): boolean {
+ if (!(toolNameMap instanceof Map)) return false;
+ if (!parsed || typeof parsed !== "object") return false;
+
+ const block =
+ parsed.content_block && typeof parsed.content_block === "object"
+ ? (parsed.content_block as JsonRecord)
+ : null;
+ if (!block || block.type !== "tool_use" || typeof block.name !== "string") return false;
+
+ const restoredName = toolNameMap.get(block.name) ?? block.name;
+ if (restoredName === block.name) return false;
+ block.name = restoredName;
+ return true;
+}
\ No newline at end of file
diff --git a/open-sse/utils/stream/errors.ts b/open-sse/utils/stream/errors.ts
new file mode 100644
index 00000000000..342cc5d8374
--- /dev/null
+++ b/open-sse/utils/stream/errors.ts
@@ -0,0 +1,87 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { asRecord } from "./utils.ts";
+import { JsonRecord, StreamFailurePayload } from "./types.ts";
+
+export function toStreamFailureStatus(value: unknown): number | null {
+ if (typeof value === "number" && Number.isInteger(value) && value >= 400 && value <= 599) {
+ return value;
+ }
+ if (typeof value === "string" && /^\d{3}$/.test(value.trim())) {
+ const parsed = Number(value.trim());
+ return parsed >= 400 && parsed <= 599 ? parsed : null;
+ }
+ return null;
+}
+
+export function looksLikeStreamRateLimit(code: string, type: string, message: string): boolean {
+ const haystack = `${code} ${type} ${message}`.toLowerCase();
+ return (
+ haystack.includes("usage_limit_reached") ||
+ haystack.includes("rate_limit") ||
+ haystack.includes("rate limit") ||
+ haystack.includes("quota") ||
+ haystack.includes("too many requests") ||
+ haystack.includes("limit reached") ||
+ haystack.includes("limit has been reached")
+ );
+}
+
+function resolveErrorSource(response: JsonRecord, record: JsonRecord): JsonRecord {
+ const responseError = asRecord(response.error);
+ if (Object.keys(responseError).length) return responseError;
+ const recordError = asRecord(record.error);
+ if (Object.keys(recordError).length) return recordError;
+ return record;
+}
+
+function resolveStreamFailureMessage(error: JsonRecord, record: JsonRecord): string {
+ if (typeof error.message === "string" && error.message.trim()) {
+ return error.message;
+ }
+ if (typeof record.message === "string" && record.message.trim()) {
+ return record.message;
+ }
+ return "Upstream failure";
+}
+
+function resolveStreamFailureStatus(
+ error: JsonRecord,
+ response: JsonRecord,
+ record: JsonRecord,
+ code: string,
+ type: string | undefined,
+ message: string
+): number {
+ const candidates: unknown[] = [
+ error.status_code,
+ error.status,
+ response.status_code,
+ response.status,
+ record.status_code,
+ record.status,
+ ];
+ for (const candidate of candidates) {
+ const result = toStreamFailureStatus(candidate);
+ if (result !== null) return result;
+ }
+ return looksLikeStreamRateLimit(code, type || "", message) ? 429 : 502;
+}
+
+export function normalizeStreamFailurePayload(payload: unknown): StreamFailurePayload | null {
+ const record = payload && typeof payload === "object" ? (payload as JsonRecord) : {};
+ const response = asRecord(record.response);
+ const error = resolveErrorSource(response, record);
+ const code = typeof error.code === "string" ? error.code : "upstream_error";
+ const type = typeof error.type === "string" ? error.type : undefined;
+ const message = resolveStreamFailureMessage(error, record);
+ const status = resolveStreamFailureStatus(error, response, record, code, type, message);
+
+ return {
+ status,
+ message,
+ code,
+ ...(type ? { type } : {}),
+ };
+}
diff --git a/open-sse/utils/stream/index.ts b/open-sse/utils/stream/index.ts
new file mode 100644
index 00000000000..f22182abe95
--- /dev/null
+++ b/open-sse/utils/stream/index.ts
@@ -0,0 +1,9 @@
+export * from "./types.ts";
+export * from "./utils.ts";
+export * from "./responsesLifecycle.ts";
+export * from "./textualToolCalls.ts";
+export * from "./sseFormatters.ts";
+export * from "./errors.ts";
+export * from "./claudeLifecycle.ts";
+export * from "./openaiChunks.ts";
+export * from "./streamCore.ts";
diff --git a/open-sse/utils/stream/openaiChunks.ts b/open-sse/utils/stream/openaiChunks.ts
new file mode 100644
index 00000000000..4c0b3a9c426
--- /dev/null
+++ b/open-sse/utils/stream/openaiChunks.ts
@@ -0,0 +1,10 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { JsonRecord } from "./types.ts";
+
+export function getOpenAIIntermediateChunks(value: unknown): unknown[] {
+ if (!value || typeof value !== "object") return [];
+ const candidate = (value as JsonRecord)._openaiIntermediate;
+ return Array.isArray(candidate) ? candidate : [];
+}
\ No newline at end of file
diff --git a/open-sse/utils/stream/responsesLifecycle.ts b/open-sse/utils/stream/responsesLifecycle.ts
new file mode 100644
index 00000000000..1a37401a952
--- /dev/null
+++ b/open-sse/utils/stream/responsesLifecycle.ts
@@ -0,0 +1,203 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { stringifyIdValue } from "./utils.ts";
+import { JsonRecord } from "./types.ts";
+
+function isNonArrayRecord(value: unknown): value is Record {
+ return typeof value === "object" && value !== null && !Array.isArray(value);
+}
+
+export function normalizeResponsesOutputItemIds(item: unknown): unknown {
+ if (!item || typeof item !== "object" || Array.isArray(item)) {
+ return item;
+ }
+
+ const record = item as JsonRecord;
+ let changed = false;
+ const normalized = { ...record };
+
+ const id = stringifyIdValue(record.id);
+ if (id !== null && record.id !== id) {
+ normalized.id = id;
+ changed = true;
+ }
+
+ const callId = stringifyIdValue(record.call_id);
+ if (callId !== null && record.call_id !== callId) {
+ normalized.call_id = callId;
+ changed = true;
+ }
+
+ return changed ? normalized : item;
+}
+
+export function normalizeResponsesSseIds(payload: JsonRecord): boolean {
+ let changed = false;
+
+ for (const key of ["response_id", "item_id", "call_id"] as const) {
+ const value = stringifyIdValue(payload[key]);
+ if (value !== null && payload[key] !== value) {
+ payload[key] = value;
+ changed = true;
+ }
+ }
+
+ if (isNonArrayRecord(payload.item)) {
+ const normalizedItem = normalizeResponsesOutputItemIds(payload.item);
+ if (normalizedItem !== payload.item) {
+ payload.item = normalizedItem;
+ changed = true;
+ }
+ }
+
+ if (isNonArrayRecord(payload.response)) {
+ const response = payload.response as JsonRecord;
+ let responseChanged = false;
+ const normalizedResponse = { ...response };
+
+ const responseId = stringifyIdValue(response.id);
+ if (responseId !== null && response.id !== responseId) {
+ normalizedResponse.id = responseId;
+ responseChanged = true;
+ }
+
+ if (Array.isArray(response.output)) {
+ const normalizedOutput = response.output.map(normalizeResponsesOutputItemIds);
+ if (normalizedOutput.some((item, index) => item !== response.output[index])) {
+ normalizedResponse.output = normalizedOutput;
+ responseChanged = true;
+ }
+ }
+
+ if (responseChanged) {
+ payload.response = normalizedResponse;
+ changed = true;
+ }
+ }
+
+ return changed;
+}
+
+export const PENDING_REQUEST_CLEARED_MARKER = "__omniroutePendingRequestCleared";
+
+export function markPendingRequestCleared(error: Error): Error {
+ (error as Error & Record)[PENDING_REQUEST_CLEARED_MARKER] = true;
+ return error;
+}
+
+export function buildResponsesOutputItemKey(item: unknown): string | null {
+ if (!item || typeof item !== "object" || Array.isArray(item)) {
+ return null;
+ }
+
+ const record = item as JsonRecord;
+ const type = typeof record.type === "string" ? record.type : "";
+ const id = stringifyIdValue(record.id) ?? "";
+ const callId = stringifyIdValue(record.call_id) ?? "";
+ const outputIndex = typeof record.output_index === "number" ? record.output_index : "";
+ const name = typeof record.name === "string" ? record.name : "";
+
+ if (!type && !id && !callId) {
+ return null;
+ }
+
+ return `${type}:${id}:${callId}:${outputIndex}:${name}`;
+}
+
+export function pushUniqueResponsesOutputItems(target: unknown[], items: readonly unknown[]) {
+ const seen = new Set();
+
+ for (const existingItem of target) {
+ const key = buildResponsesOutputItemKey(existingItem);
+ if (key) {
+ seen.add(key);
+ }
+ }
+
+ for (const item of items) {
+ const key = buildResponsesOutputItemKey(item);
+ if (key && seen.has(key)) {
+ continue;
+ }
+
+ target.push(item);
+ if (key) {
+ seen.add(key);
+ }
+ }
+}
+
+/**
+ * Lifecycle event types in OpenAI Responses API streams whose `response`
+ * payload is a snapshot of the request (echoes back `instructions` + `tools`).
+ */
+export const RESPONSES_LIFECYCLE_EVENT_TYPES = new Set([
+ "response.created",
+ "response.in_progress",
+ "response.completed",
+]);
+
+/**
+ * Backfill `parsed.response.output` on a `response.completed` event from the
+ * snapshots accumulated as the stream progressed (`response.output_item.done`).
+ *
+ * Why: when the upstream request runs with `store: false`, OpenAI's Responses
+ * API leaves `response.output` empty in the final `response.completed`
+ * snapshot — clients that rebuild assistant messages from that snapshot
+ * (notably the GitHub Copilot CLI 1.0.36) end up with `choices: []` and never
+ * trigger tool execution. Codex CLI and others that consume per-item events
+ * are unaffected; backfilling the array makes both styles work.
+ *
+ * Returns true when `parsed.response.output` was empty and got replaced, so
+ * the caller can re-serialize.
+ */
+export function backfillResponsesCompletedOutput(
+ parsed: unknown,
+ collectedItems: readonly unknown[]
+): boolean {
+ if (!collectedItems.length) return false;
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
+ const obj = parsed as Record;
+ if (obj.type !== "response.completed") return false;
+ const resp = obj.response;
+ if (!resp || typeof resp !== "object" || Array.isArray(resp)) return false;
+ const r = resp as Record;
+ const existing = r.output;
+ if (Array.isArray(existing) && existing.length > 0) return false;
+ r.output = collectedItems.slice();
+ return true;
+}
+
+/**
+ * Strip the request echo (`instructions`, `tools`) from `parsed.response`
+ * on Responses API lifecycle events.
+ *
+ * Why: those fields can balloon the SSE message past 100 KB when the request
+ * carries large tool definitions / instructions. Some clients (notably the
+ * GitHub Copilot CLI) cannot process oversized SSE events and stop rendering
+ * mid-stream. The fields are pure echo of the original request — clients
+ * already hold the original locally — so removing them is observably safe.
+ *
+ * Returns true when the payload was modified and must be re-serialized.
+ */
+export function stripResponsesLifecycleEcho(parsed: unknown): boolean {
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
+ const obj = parsed as Record;
+ if (typeof obj.type !== "string" || !RESPONSES_LIFECYCLE_EVENT_TYPES.has(obj.type)) {
+ return false;
+ }
+ const resp = obj.response;
+ if (!resp || typeof resp !== "object" || Array.isArray(resp)) return false;
+ const r = resp as Record;
+ let changed = false;
+ if ("instructions" in r) {
+ delete r.instructions;
+ changed = true;
+ }
+ if ("tools" in r) {
+ delete r.tools;
+ changed = true;
+ }
+ return changed;
+}
diff --git a/open-sse/utils/stream/sseFormatters.ts b/open-sse/utils/stream/sseFormatters.ts
new file mode 100644
index 00000000000..0d2c759118b
--- /dev/null
+++ b/open-sse/utils/stream/sseFormatters.ts
@@ -0,0 +1,90 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { asRecord } from "./utils.ts";
+import { JsonRecord, ToolCall } from "./types.ts";
+
+/* @testonly */ export function toStreamingToolCallDelta(toolCall: ToolCall) {
+ return {
+ index: toolCall.index,
+ id: toolCall.id != null ? String(toolCall.id) : null,
+ type: toolCall.type,
+ function: {
+ name: toolCall.function.name,
+ arguments: toolCall.function.arguments,
+ },
+ };
+}
+
+/* @testonly */ export function toResponsesFunctionCallItem(toolCall: ToolCall) {
+ return {
+ type: "function_call",
+ id: (toolCall.id != null ? String(toolCall.id) : null) || `fc_${toolCall.index}`,
+ call_id: (toolCall.id != null ? String(toolCall.id) : null) || `call_${toolCall.index}`,
+ name: toolCall.function.name,
+ arguments: toolCall.function.arguments,
+ status: "completed",
+ };
+}
+
+export function buildResponsesFunctionCallEvents(toolCall: ToolCall) {
+ const item = toResponsesFunctionCallItem(toolCall);
+ return [
+ {
+ type: "response.output_item.added",
+ output_index: toolCall.index,
+ item,
+ },
+ {
+ type: "response.function_call_arguments.done",
+ item_id: item.id,
+ output_index: toolCall.index,
+ arguments: toolCall.function.arguments,
+ },
+ {
+ type: "response.output_item.done",
+ output_index: toolCall.index,
+ item,
+ },
+ ];
+}
+
+export function formatSSEDataEvents(events: unknown[]) {
+ return events.map((event) => `data: ${JSON.stringify(event)}\n`).join("\n");
+}
+
+export function toChatCompletionChunkWithToolCall(base: JsonRecord, toolCall: ToolCall) {
+ const choice = asRecord(Array.isArray(base.choices) ? base.choices[0] : null);
+ const delta = { ...asRecord(choice.delta) };
+ delete delta.content;
+ delete delta.reasoning_content;
+ return {
+ ...base,
+ choices: [
+ {
+ ...choice,
+ index: typeof choice.index === "number" ? choice.index : 0,
+ delta: {
+ ...delta,
+ tool_calls: [toStreamingToolCallDelta(toolCall)],
+ },
+ finish_reason: null,
+ },
+ ],
+ };
+}
+
+export function toResponsesCompletedWithToolCalls(parsed: JsonRecord, toolCalls: ToolCall[]) {
+ const response = asRecord(parsed.response);
+ const existingOutput = Array.isArray(response.output) ? response.output : [];
+ return {
+ ...parsed,
+ response: {
+ ...response,
+ output: [
+ ...existingOutput,
+ ...toolCalls.map((toolCall) => toResponsesFunctionCallItem(toolCall)),
+ ],
+ },
+ };
+}
\ No newline at end of file
diff --git a/open-sse/utils/stream/streamCore.ts b/open-sse/utils/stream/streamCore.ts
new file mode 100644
index 00000000000..2c736ab45d0
--- /dev/null
+++ b/open-sse/utils/stream/streamCore.ts
@@ -0,0 +1,2215 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { translateResponse, initState } from "../../translator/index.ts";
+import { v4 as uuidv4 } from "uuid";
+import { FORMATS } from "../../translator/formats.ts";
+import { generateSessionId } from "../../services/sessionManager.ts";
+import { trackPendingRequest, appendRequestLog } from "@/lib/usage/usageHistory.ts";
+import { calculateCost } from "@/lib/usage/costCalculator";
+import { buildOmniRouteSseMetadataComment } from "@/domain/omnirouteResponseMeta";
+import { createStructuredSSECollector } from "../streamPayloadCollector.ts";
+import {
+ STREAM_IDLE_TIMEOUT_MS,
+ FETCH_BODY_TIMEOUT_MS,
+ HTTP_STATUS,
+} from "../../config/constants.ts";
+import { parseSSELine } from "../streamHelpers.ts";
+import { recordToolLatency } from "../../services/toolLatencyTracker.ts";
+import { processBufferedPassthroughLine } from "../passthroughTailProcessor.ts";
+import { normalizeStreamFailurePayload } from "./errors.ts";
+import { buildErrorBody } from "../error.ts";
+import { SSEStreamContext } from "./types.ts";
+
+import {
+ parseTextualToolCallFromContent,
+ containsTextualToolCallCandidate,
+ containsMalformedTextualToolCall,
+ extractAllowedToolNames,
+ collectPassthroughTextualToolCall,
+} from "./textualToolCalls.ts";
+import { getOpenAIIntermediateChunks } from "./openaiChunks.ts";
+import {
+ SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT,
+ createClaudeEmptyResponseLifecycle,
+ getClaudeEventType,
+ isClaudeEventPayload,
+ updateClaudeEmptyResponseLifecycle,
+ shouldInjectClaudeEmptyResponseBeforeCurrentEvent,
+ shouldInjectClaudeEmptyResponseOnFlush,
+ shouldInjectClaudeMissingFinalizersOnFlush,
+ buildSyntheticClaudeEmptyResponseEvents,
+ restoreClaudePassthroughToolUseName,
+} from "./claudeLifecycle.ts";
+import {
+ normalizeResponsesSseIds,
+ markPendingRequestCleared,
+ pushUniqueResponsesOutputItems,
+ backfillResponsesCompletedOutput,
+ stripResponsesLifecycleEcho,
+} from "./responsesLifecycle.ts";
+import {
+ buildResponsesFunctionCallEvents,
+ formatSSEDataEvents,
+ toChatCompletionChunkWithToolCall,
+ toResponsesCompletedWithToolCalls,
+} from "./sseFormatters.ts";
+import { stringifyIdValue, asRecord, appendBoundedText, STREAM_MODE } from "./utils.ts";
+import {
+ JsonRecord,
+ StreamLogger,
+ StreamCompletePayload,
+ StreamFailurePayload,
+ StreamOptions,
+ TranslateState,
+ ToolCall,
+ UsageTokenRecord,
+} from "./types.ts";
+import { normalizeStreamFailurePayload } from "./errors.ts";
+import { buildErrorBody } from "../error.ts";
+
+// Module-level helpers extracted from createSSEStream closures to keep
+// cyclomatic complexity of individual functions under the 15-node gate.
+
+function getResponsesReasoningSummaryTextModule(item: Record): string {
+ return Array.isArray(item.summary)
+ ? item.summary
+ .map((part) => {
+ if (!part || typeof part !== "object" || Array.isArray(part)) {
+ return "";
+ }
+ return typeof (part as Record).text === "string"
+ ? ((part as Record).text as string)
+ : "";
+ })
+ .join("")
+ : "";
+}
+
+function getResponsesReasoningKeyModule(
+ payload: Record,
+ passthroughResponsesId: string | null
+): string | null {
+ const itemId = stringifyIdValue(payload.item_id);
+ if (itemId) {
+ return itemId;
+ }
+ const item =
+ payload.item && typeof payload.item === "object" && !Array.isArray(payload.item)
+ ? (payload.item as Record)
+ : null;
+ const outputItemId = item ? stringifyIdValue(item.id) : null;
+ if (outputItemId) {
+ return outputItemId;
+ }
+ const responseId = stringifyIdValue(payload.response_id) || passthroughResponsesId;
+ const outputIndex =
+ typeof payload.output_index === "number" && Number.isInteger(payload.output_index)
+ ? payload.output_index
+ : null;
+ return responseId !== null && outputIndex !== null ? `${responseId}:${outputIndex}` : null;
+}
+
+function ensureVisibleResponsesReasoningSummaryModule(payload: Record): boolean {
+ const item =
+ payload.item && typeof payload.item === "object" && !Array.isArray(payload.item)
+ ? (payload.item as Record)
+ : null;
+ if (!item || item.type !== "reasoning") {
+ return false;
+ }
+ if (getResponsesReasoningSummaryTextModule(item)) {
+ return false;
+ }
+ const hasEncryptedReasoning =
+ typeof item.encrypted_content === "string" && item.encrypted_content.length > 0;
+ if (!hasEncryptedReasoning) {
+ return false;
+ }
+ item.summary = [
+ {
+ type: "summary_text",
+ text: "Codex is reasoning, but the upstream Responses API exposed this reasoning block only as encrypted state. OmniRoute cannot recover the private reasoning text.",
+ },
+ ];
+ return true;
+}
+
+function computeReasoningIdsModule(
+ item: Record,
+ payload: Record,
+ reasoningKey: string
+): { itemId: string; outputIndex: number } {
+ const itemId = typeof item.id === "string" && item.id ? item.id : reasoningKey;
+ const outputIndex =
+ typeof payload.output_index === "number" && Number.isInteger(payload.output_index)
+ ? payload.output_index
+ : 0;
+ return { itemId, outputIndex };
+}
+
+function maybeExtractOpenAIThinking(
+ itemSanitized: Record,
+ sourceFormat: string
+): Record | null {
+ const isResponsesEvent =
+ typeof itemSanitized?.event === "string" && itemSanitized.event.startsWith("response.");
+ if (sourceFormat === FORMATS.OPENAI && !isResponsesEvent) {
+ const sanitized = sanitizeStreamingChunk(itemSanitized) as Record;
+ const delta = sanitized?.choices?.[0]?.delta;
+ if (delta?.content && typeof delta.content === "string") {
+ const { content, thinking } = extractThinkingFromContent(delta.content);
+ delta.content = content;
+ if (thinking && !delta.reasoning_content) {
+ delta.reasoning_content = thinking;
+ }
+ }
+ return sanitized;
+ }
+ return null;
+}
+
+function maybeFillFinishChunkUsage(
+ itemSanitized: Record,
+ state: TranslateState | null | undefined,
+ isFinishChunk: boolean,
+ body: Record | null | undefined,
+ totalContentLength: number,
+ sourceFormat: string
+): void {
+ if (!state?.finishReason || !isFinishChunk) {
+ return;
+ }
+ if (!hasValidUsage(itemSanitized.usage) && totalContentLength > 0) {
+ const estimated = estimateUsage(body, totalContentLength, sourceFormat);
+ itemSanitized.usage = filterUsageForFormat(estimated, sourceFormat);
+ state.usage = estimated;
+ } else if (state.usage) {
+ const buffered = addBufferToUsage(state.usage);
+ itemSanitized.usage = filterUsageForFormat(buffered, sourceFormat);
+ }
+}
+
+function maybeEmitClaudeLifecycleEvents(
+ itemSanitized: Record,
+ claudeEmptyResponseLifecycle: Record,
+ sourceFormat: string
+): { shouldInjectEmptyResponse: boolean; eventType: string | null } {
+ let shouldInjectEmptyResponse = false;
+ let eventType: string | null = null;
+ if (
+ sourceFormat === FORMATS.CLAUDE &&
+ shouldInjectClaudeEmptyResponseBeforeCurrentEvent(
+ claudeEmptyResponseLifecycle as Parameters<
+ typeof shouldInjectClaudeEmptyResponseBeforeCurrentEvent
+ >[0],
+ itemSanitized
+ )
+ ) {
+ shouldInjectEmptyResponse = true;
+ eventType = getClaudeEventType(itemSanitized);
+ }
+ if (sourceFormat === FORMATS.CLAUDE && isClaudeEventPayload(itemSanitized)) {
+ updateClaudeEmptyResponseLifecycle(
+ claudeEmptyResponseLifecycle as Parameters[0],
+ itemSanitized
+ );
+ }
+ return { shouldInjectEmptyResponse, eventType };
+}
+
+const CREATE_SSE_STREAM_DEFAULTS = {
+ mode: STREAM_MODE.TRANSLATE,
+ clientResponseFormat: null,
+ copilotCompatibleReasoning: false,
+ provider: null,
+ reqLogger: null,
+ toolNameMap: null,
+ model: null,
+ connectionId: null,
+ apiKeyInfo: null,
+ body: null,
+ onComplete: null,
+ onFailure: null,
+};
+
+function buildSSEStreamContext(options: StreamOptions): SSEStreamContext {
+ const {
+ mode,
+ targetFormat,
+ sourceFormat,
+ clientResponseFormat,
+ copilotCompatibleReasoning,
+ provider,
+ reqLogger,
+ toolNameMap,
+ model,
+ connectionId,
+ apiKeyInfo,
+ body,
+ onComplete,
+ onFailure,
+ } = { ...CREATE_SSE_STREAM_DEFAULTS, ...options };
+ const signatureNamespace = connectionId;
+
+ const clientExpectsResponsesStream =
+ (mode === STREAM_MODE.PASSTHROUGH
+ ? clientResponseFormat === FORMATS.OPENAI_RESPONSES
+ : sourceFormat === FORMATS.OPENAI_RESPONSES) === true;
+
+ const clientExpectsClaudeStream =
+ (mode === STREAM_MODE.PASSTHROUGH
+ ? clientResponseFormat === FORMATS.CLAUDE
+ : sourceFormat === FORMATS.CLAUDE) === true;
+
+ const shouldEmitDoneTerminator = !clientExpectsResponsesStream && !clientExpectsClaudeStream;
+
+ let buffer = "";
+ let usage: UsageTokenRecord | null = null;
+ let passthroughHasToolCalls = false;
+ const passthroughToolCalls = new Map();
+ let passthroughToolCallSeq = 0;
+ const allowedToolNames = extractAllowedToolNames(body);
+ let skipPassthroughEvent = false;
+
+ const state: TranslateState | null =
+ mode === STREAM_MODE.TRANSLATE
+ ? {
+ ...(initState(sourceFormat) as TranslateState),
+ provider,
+ toolNameMap,
+ signatureNamespace,
+ copilotCompatibleReasoning,
+ accumulatedContent: "",
+ }
+ : null;
+
+ let totalContentLength = 0;
+ let passthroughAccumulatedContent = "";
+ let passthroughAccumulatedReasoning = "";
+ let passthroughBufferedTextualToolCallContent = "";
+ const passthroughResponsesOutputItems: unknown[] = [];
+ const passthroughResponsesPendingFunctionCalls = new Map();
+ let passthroughResponsesId: string | null = null;
+ let passthroughResponsesCurrentFunctionCallKey: string | null = null;
+ const passthroughResponsesReasoningSummarySeen = new Set();
+ const streamStartedAt = Date.now();
+
+ let lastToolCallChunkTime: number | null = null;
+ let toolFinishTime: number | null = null;
+ let contentAfterToolSeen = false;
+
+ const sessionId = generateSessionId(body as Parameters[0], {
+ provider: provider ?? undefined,
+ connectionId: connectionId ?? undefined,
+ });
+ let pendingToolFinishTime: number | null = null;
+ try {
+ pendingToolFinishTime = consumeToolFinishTime(sessionId);
+ } catch {}
+
+ let doneSent = false;
+ const providerPayloadCollector = createStructuredSSECollector({
+ stage: "provider_response",
+ });
+ const clientPayloadCollector = createStructuredSSECollector({
+ stage: "client_response",
+ });
+ const requestRecord = asRecord(body);
+ const requestStreamOptions = asRecord(
+ requestRecord.stream_options ?? requestRecord.streamOptions
+ );
+ const expectsOpenAIUsageOnlyChunk =
+ requestStreamOptions.include_usage === true || requestStreamOptions.includeUsage === true;
+
+ const decoder = new TextDecoder();
+ const encoder = new TextEncoder();
+
+ let lastChunkTime = Date.now();
+ let idleTimer: ReturnType | null = null;
+ let streamTimedOut = false;
+ const claudeEmptyResponseLifecycle = createClaudeEmptyResponseLifecycle() as Record<
+ string,
+ unknown
+ >;
+ let pendingPassthroughEventLine: string | null = null;
+ let pendingPassthroughEventEmitted = false;
+
+ const clearIdleTimer = () => {
+ if (idleTimer) {
+ clearInterval(idleTimer);
+ idleTimer = null;
+ }
+ };
+
+ const clearPendingPassthroughEvent = () => {
+ pendingPassthroughEventLine = null;
+ pendingPassthroughEventEmitted = false;
+ };
+
+ const maybePrefixPendingPassthroughEvent = (output: string, line: string) => {
+ if (!pendingPassthroughEventLine || !line.startsWith("data:")) {
+ return output;
+ }
+ if (!pendingPassthroughEventEmitted) {
+ pendingPassthroughEventEmitted = true;
+ return `${pendingPassthroughEventLine}\n${output}`;
+ }
+ return output;
+ };
+
+ const applyTextualToolCallStreamingGuard = (parsed: Record) => {
+ const choice = Array.isArray((parsed as JsonRecord).choices)
+ ? (((parsed as JsonRecord).choices as unknown[])[0] as JsonRecord | undefined)
+ : undefined;
+ const delta = asRecord(choice?.delta);
+ let textualToolCallConverted = false;
+
+ if (typeof delta?.content === "string") {
+ const incomingContent = delta.content;
+ const bufferedCandidate = passthroughBufferedTextualToolCallContent + incomingContent;
+ if (
+ passthroughBufferedTextualToolCallContent ||
+ containsTextualToolCallCandidate(incomingContent)
+ ) {
+ const parsedCandidate = parseTextualToolCallCandidate(bufferedCandidate);
+ if (parsedCandidate?.kind === "complete") {
+ const collectedToolCall = collectPassthroughTextualToolCall(
+ bufferedCandidate,
+ passthroughToolCalls,
+ allowedToolNames
+ );
+ if (collectedToolCall) {
+ parsed = toChatCompletionChunkWithToolCall(parsed, collectedToolCall);
+ passthroughHasToolCalls = true;
+ } else {
+ delete delta.content;
+ delete delta.reasoning_content;
+ }
+ textualToolCallConverted = true;
+ passthroughBufferedTextualToolCallContent = "";
+ } else if (parsedCandidate?.kind === "partial") {
+ passthroughBufferedTextualToolCallContent = appendBoundedText(
+ passthroughBufferedTextualToolCallContent,
+ incomingContent
+ );
+ textualToolCallConverted = true;
+ delta.content = "";
+ } else {
+ if (passthroughBufferedTextualToolCallContent) {
+ delta.content = passthroughBufferedTextualToolCallContent + incomingContent;
+ textualToolCallConverted = true;
+ }
+ passthroughAccumulatedContent = appendBoundedText(
+ passthroughAccumulatedContent,
+ passthroughBufferedTextualToolCallContent + incomingContent
+ );
+ passthroughBufferedTextualToolCallContent = "";
+ }
+ } else {
+ passthroughAccumulatedContent = appendBoundedText(
+ passthroughAccumulatedContent,
+ incomingContent
+ );
+ }
+ }
+
+ return { parsed, textualToolCallConverted };
+ };
+
+ const emitSyntheticClaudeEmptyResponse = (
+ controller: TransformStreamDefaultController,
+ options: {
+ includeContentBlock?: boolean;
+ includeMessageDelta?: boolean;
+ includeMessageStop?: boolean;
+ } = {}
+ ) => {
+ const events = buildSyntheticClaudeEmptyResponseEvents(
+ claudeEmptyResponseLifecycle,
+ model,
+ options
+ );
+ if (events.length === 0) return;
+
+ if (!claudeEmptyResponseLifecycle.warningLogged) {
+ claudeEmptyResponseLifecycle.warningLogged = true;
+ console.warn(
+ `[STREAM] Injecting synthetic Claude SSE response for empty upstream output (${provider || "provider"}:${model || "unknown"})`
+ );
+ }
+
+ if (options.includeContentBlock !== false) {
+ claudeEmptyResponseLifecycle.syntheticContentInjected = true;
+ if (!passthroughAccumulatedContent.trim()) {
+ passthroughAccumulatedContent = SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT;
+ }
+ if (state?.accumulatedContent !== undefined && !state.accumulatedContent.trim()) {
+ state.accumulatedContent = SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT;
+ }
+ }
+
+ for (const event of events) {
+ updateClaudeEmptyResponseLifecycle(claudeEmptyResponseLifecycle, event);
+ clientPayloadCollector.push(event);
+ const output = formatSSE(event, FORMATS.CLAUDE);
+ reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(encoder.encode(output));
+ }
+ };
+
+ const emitTranslatedClientItem = (
+ controller: TransformStreamDefaultController,
+ item: Record
+ ) => {
+ let itemSanitized: Record = item;
+
+ const sanitized = maybeExtractOpenAIThinking(itemSanitized, sourceFormat);
+ if (sanitized) {
+ itemSanitized = sanitized;
+ }
+
+ if (!hasValuableContent(itemSanitized, sourceFormat)) {
+ return;
+ }
+
+ const isFinishChunk =
+ itemSanitized.type === "message_delta" || itemSanitized.choices?.[0]?.finish_reason;
+ maybeFillFinishChunkUsage(
+ itemSanitized,
+ state,
+ isFinishChunk,
+ body,
+ totalContentLength,
+ sourceFormat
+ );
+
+ const { shouldInjectEmptyResponse, eventType } = maybeEmitClaudeLifecycleEvents(
+ itemSanitized,
+ claudeEmptyResponseLifecycle,
+ sourceFormat
+ );
+ if (shouldInjectEmptyResponse) {
+ emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: true,
+ includeMessageDelta:
+ eventType === "message_stop" &&
+ !(claudeEmptyResponseLifecycle as Record).hasMessageDelta,
+ includeMessageStop: false,
+ });
+ }
+
+ const output = formatSSE(itemSanitized, sourceFormat);
+ clientPayloadCollector.push(itemSanitized);
+ reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(encoder.encode(output));
+ };
+
+ const emitFinalSseMetadata = async (
+ controller: TransformStreamDefaultController,
+ finalUsage: UsageTokenRecord | Record | null | undefined
+ ) => {
+ const costUsd = finalUsage ? await calculateCost(provider, model, finalUsage) : 0;
+ const comment = buildOmniRouteSseMetadataComment({
+ provider,
+ model,
+ cacheHit: false,
+ latencyMs: Date.now() - streamStartedAt,
+ usage: finalUsage,
+ costUsd,
+ });
+ if (!comment) return;
+ reqLogger?.appendConvertedChunk?.(comment);
+ controller.enqueue(encoder.encode(comment));
+ };
+
+ const getResponsesReasoningKey = (payload: Record): string | null =>
+ getResponsesReasoningKeyModule(payload, passthroughResponsesId);
+
+ const getResponsesReasoningSummaryText = getResponsesReasoningSummaryTextModule;
+
+ const ensureVisibleResponsesReasoningSummary = ensureVisibleResponsesReasoningSummaryModule;
+
+ const emitSyntheticResponsesReasoningSummary = (
+ controller: TransformStreamDefaultController,
+ payload: Record
+ ) => {
+ const item =
+ payload.item && typeof payload.item === "object" && !Array.isArray(payload.item)
+ ? (payload.item as Record)
+ : null;
+ if (!item || item.type !== "reasoning") {
+ return;
+ }
+
+ ensureVisibleResponsesReasoningSummary(payload);
+ const visibleSummary = getResponsesReasoningSummaryText(item);
+
+ if (!visibleSummary) {
+ return;
+ }
+
+ const reasoningKey = getResponsesReasoningKey(payload);
+ if (!reasoningKey || passthroughResponsesReasoningSummarySeen.has(reasoningKey)) {
+ return;
+ }
+ passthroughResponsesReasoningSummarySeen.add(reasoningKey);
+
+ const { itemId, outputIndex } = computeReasoningIdsModule(item, payload, reasoningKey);
+
+ const syntheticEvents = [
+ {
+ event: "response.reasoning_summary_text.delta",
+ body: {
+ type: "response.reasoning_summary_text.delta",
+ item_id: itemId,
+ output_index: outputIndex,
+ summary_index: 0,
+ delta: visibleSummary,
+ },
+ },
+ {
+ event: "response.reasoning_summary_part.done",
+ body: {
+ type: "response.reasoning_summary_part.done",
+ item_id: itemId,
+ output_index: outputIndex,
+ summary_index: 0,
+ part: { type: "summary_text", text: visibleSummary },
+ },
+ },
+ ];
+
+ for (const syntheticEvent of syntheticEvents) {
+ clientPayloadCollector.push(syntheticEvent.body);
+ const output = `event: ${syntheticEvent.event}\ndata: ${JSON.stringify(syntheticEvent.body)}\n\n`;
+ reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(encoder.encode(output));
+ }
+ };
+
+ return {
+ mode: mode as string,
+ targetFormat: targetFormat as string | undefined,
+ sourceFormat: sourceFormat as string | undefined,
+ clientResponseFormat: clientResponseFormat as string | null,
+ copilotCompatibleReasoning: copilotCompatibleReasoning as boolean,
+ provider: provider as string | null,
+ reqLogger: reqLogger as StreamLogger | null,
+ toolNameMap,
+ model: model as string | null,
+ connectionId: connectionId as string | null,
+ apiKeyInfo,
+ body,
+ onComplete: onComplete as ((payload: StreamCompletePayload) => void) | null,
+ onFailure: onFailure as ((payload: StreamFailurePayload) => void | Promise) | null,
+ clientExpectsResponsesStream,
+ clientExpectsClaudeStream,
+ shouldEmitDoneTerminator,
+ expectsOpenAIUsageOnlyChunk,
+ signatureNamespace: signatureNamespace as string | null,
+ buffer,
+ usage,
+ passthroughHasToolCalls,
+ passthroughToolCalls,
+ passthroughToolCallSeq,
+ allowedToolNames,
+ skipPassthroughEvent,
+ state,
+ totalContentLength,
+ passthroughAccumulatedContent,
+ passthroughAccumulatedReasoning,
+ passthroughBufferedTextualToolCallContent,
+ passthroughResponsesOutputItems,
+ passthroughResponsesPendingFunctionCalls,
+ passthroughResponsesId,
+ passthroughResponsesCurrentFunctionCallKey,
+ passthroughResponsesReasoningSummarySeen,
+ streamStartedAt,
+ lastToolCallChunkTime,
+ toolFinishTime,
+ contentAfterToolSeen,
+ sessionId,
+ pendingToolFinishTime,
+ doneSent,
+ pendingPassthroughEventLine,
+ pendingPassthroughEventEmitted,
+ lastChunkTime,
+ streamTimedOut,
+ decoder,
+ encoder,
+ idleTimer,
+ claudeEmptyResponseLifecycle,
+ providerPayloadCollector,
+ clientPayloadCollector,
+ requestRecord,
+ requestStreamOptions,
+ clearIdleTimer,
+ clearPendingPassthroughEvent,
+ maybePrefixPendingPassthroughEvent,
+ applyTextualToolCallStreamingGuard,
+ emitSyntheticClaudeEmptyResponse,
+ emitTranslatedClientItem,
+ emitFinalSseMetadata,
+ getResponsesReasoningKey,
+ getResponsesReasoningSummaryText,
+ ensureVisibleResponsesReasoningSummary,
+ emitSyntheticResponsesReasoningSummary,
+ };
+}
+
+/**
+ * Create unified SSE transform stream with idle timeout protection.
+ * If the upstream provider stops sending data for STREAM_IDLE_TIMEOUT_MS,
+ * the stream emits an error event and closes to prevent indefinite hanging.
+ *
+ * @param {object} options
+ * @param {string} options.mode - Stream mode: translate, passthrough
+ * @param {string} options.targetFormat - Provider format (for translate mode)
+ * @param {string} options.sourceFormat - Client format (for translate mode)
+ * @param {string} options.provider - Provider name
+ * @param {object} options.reqLogger - Request logger instance
+ * @param {string} options.model - Model name
+ * @param {string} options.connectionId - Connection ID for usage tracking
+ * @param {object|null} options.apiKeyInfo - API key metadata for usage attribution
+ * @param {object} options.body - Request body (for input token estimation)
+ * @param {function} options.onComplete - Callback when stream finishes: ({ status, usage }) => void
+ */
+export function createSSEStream(options: StreamOptions = {}) {
+ const ctx = buildSSEStreamContext(options);
+ return new TransformStream(
+ {
+ start(controller) {
+ // Start idle watchdog — checks every 10s if ctx.provider has stopped sending
+ if (STREAM_IDLE_TIMEOUT_MS > 0) {
+ ctx.idleTimer = setInterval(() => {
+ if (!ctx.streamTimedOut && Date.now() - ctx.lastChunkTime > STREAM_IDLE_TIMEOUT_MS) {
+ ctx.streamTimedOut = true;
+ ctx.clearIdleTimer();
+ const timeoutMsg = `[STREAM] Idle timeout: no data from ${ctx.provider || "provider"} for ${STREAM_IDLE_TIMEOUT_MS}ms (ctx.model: ${ctx.model || "unknown"})`;
+ console.warn(timeoutMsg);
+ trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false);
+ appendRequestLog({
+ model: ctx.model,
+ provider: ctx.provider,
+ connectionId: ctx.connectionId,
+ status: `FAILED ${HTTP_STATUS.GATEWAY_TIMEOUT}`,
+ }).catch(() => {});
+ const timeoutError = new Error(timeoutMsg);
+ timeoutError.name = "StreamIdleTimeoutError";
+ controller.error(markPendingRequestCleared(timeoutError));
+ }
+ }, 10_000);
+ }
+ },
+
+ transform(chunk, controller) {
+ if (ctx.streamTimedOut) return;
+ ctx.lastChunkTime = Date.now();
+ const text = ctx.decoder.decode(chunk, { stream: true });
+ ctx.buffer += text;
+ ctx.reqLogger?.appendProviderChunk?.(text);
+
+ const lines = ctx.buffer.split("\n");
+ ctx.buffer = lines.pop() || "";
+
+ for (const line of lines) {
+ const trimmed = line.trim();
+
+ // Passthrough ctx.mode: normalize and forward
+ if (ctx.mode === STREAM_MODE.PASSTHROUGH) {
+ let output: string;
+ let injectedUsage = false;
+ let clientPayload: unknown = null;
+ let failurePayload: StreamFailurePayload | null = null;
+
+ if (ctx.skipPassthroughEvent) {
+ if (!trimmed) {
+ ctx.skipPassthroughEvent = false;
+ ctx.clearPendingPassthroughEvent();
+ }
+ continue;
+ }
+
+ // Drop whole keepalive event blocks — strict OpenAI-compatible SDKs
+ // try to JSON.parse empty keepalive payloads and crash.
+ if (/^event:\s*keepalive\b/i.test(trimmed)) {
+ ctx.skipPassthroughEvent = true;
+ ctx.clearPendingPassthroughEvent();
+ continue;
+ }
+
+ if (/^event:/i.test(trimmed)) {
+ if (ctx.pendingPassthroughEventLine && !ctx.pendingPassthroughEventEmitted) {
+ const pendingOutput = `${ctx.pendingPassthroughEventLine}\n`;
+ ctx.reqLogger?.appendConvertedChunk?.(pendingOutput);
+ controller.enqueue(ctx.encoder.encode(pendingOutput));
+ }
+
+ const eventType = trimmed.replace(/^event:\s*/i, "");
+ if (
+ shouldInjectClaudeEmptyResponseBeforeCurrentEvent(
+ ctx.claudeEmptyResponseLifecycle,
+ {
+ type: eventType,
+ }
+ )
+ ) {
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: true,
+ includeMessageDelta:
+ eventType === "message_stop" &&
+ !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: false,
+ });
+ }
+
+ ctx.pendingPassthroughEventLine = line;
+ ctx.pendingPassthroughEventEmitted = false;
+ continue;
+ }
+
+ if (trimmed.startsWith("data:")) {
+ const providerPayload = parseSSELine(trimmed);
+ if (providerPayload) {
+ ctx.providerPayloadCollector.push(providerPayload);
+ if ((providerPayload as { done?: unknown }).done === true) {
+ continue;
+ }
+ }
+ }
+
+ if (trimmed.startsWith("data:") && trimmed.slice(5).trim() === "[DONE]") {
+ continue;
+ }
+
+ if (trimmed.startsWith("data:") && trimmed.slice(5).trim() !== "[DONE]") {
+ try {
+ let parsed = JSON.parse(trimmed.slice(5).trim());
+
+ // Some upstream Responses-compatible providers leak an initial Chat Completions
+ // bootstrap chunk (assistant role + empty content) before emitting proper
+ // `response.*` events. That chunk is invalid on /v1/responses and breaks strict
+ // clients like OpenCode, so drop it only for Responses-native consumers.
+ const hasActiveDeltaValue = (value: unknown): boolean => {
+ if (typeof value === "string") return value.length > 0;
+ if (Array.isArray(value))
+ return value.some((entry) => hasActiveDeltaValue(entry));
+ if (value && typeof value === "object") {
+ return Object.values(value).some((entry) => hasActiveDeltaValue(entry));
+ }
+ return value !== null && value !== undefined;
+ };
+
+ const isEmptyAssistantBootstrapChunkForResponsesClient =
+ ctx.clientExpectsResponsesStream &&
+ parsed?.object === "chat.completion.chunk" &&
+ Array.isArray(parsed?.choices) &&
+ parsed.choices.length > 0 &&
+ parsed.choices.every((choice) => {
+ const candidate = choice && typeof choice === "object" ? choice : {};
+ const delta =
+ candidate.delta && typeof candidate.delta === "object"
+ ? candidate.delta
+ : null;
+
+ if (!delta || delta.role !== "assistant") return false;
+ if (hasActiveDeltaValue(delta.content)) return false;
+ if (candidate.finish_reason !== null && candidate.finish_reason !== undefined) {
+ return false;
+ }
+
+ const { role: _role, content: _content, ...restDelta } = delta;
+ return !hasActiveDeltaValue(restDelta);
+ });
+
+ if (isEmptyAssistantBootstrapChunkForResponsesClient) {
+ continue;
+ }
+
+ // Detect Responses SSE payloads (have a `type` field like "response.created",
+ // "response.output_item.added", etc.) and skip Chat Completions-specific
+ // sanitization to avoid corrupting the stream for Responses-native clients.
+ const isResponsesSSE =
+ parsed.type &&
+ typeof parsed.type === "string" &&
+ parsed.type.startsWith("response.");
+
+ // Detect Claude SSE payloads. Includes "ping" and "error" to ensure
+ // they bypass the Chat Completions sanitization path which would
+ // incorrectly process or drop them.
+ const isClaudeSSE =
+ parsed.type &&
+ typeof parsed.type === "string" &&
+ (parsed.type.startsWith("message") ||
+ parsed.type.startsWith("content_block") ||
+ parsed.type === "ping" ||
+ parsed.type === "error");
+
+ if (isResponsesSSE) {
+ const responsesIdsNormalized = normalizeResponsesSseIds(parsed as JsonRecord);
+ const parsedResponse =
+ parsed.response &&
+ typeof parsed.response === "object" &&
+ !Array.isArray(parsed.response)
+ ? (parsed.response as JsonRecord)
+ : null;
+ const responseId =
+ (parsedResponse ? stringifyIdValue(parsedResponse.id) : null) ||
+ stringifyIdValue(parsed.response_id);
+ if (responseId) {
+ ctx.passthroughResponsesId = responseId;
+ }
+ const extracted = extractUsage(parsed);
+ if (extracted) {
+ ctx.usage = extracted;
+ }
+ if (typeof parsed.delta === "string") {
+ ctx.totalContentLength += parsed.delta.length;
+ }
+ if (
+ parsed.type === "response.output_text.delta" &&
+ typeof parsed.delta === "string"
+ ) {
+ const incomingDelta = parsed.delta;
+ const bufferedCandidate =
+ ctx.passthroughBufferedTextualToolCallContent + incomingDelta;
+ if (
+ ctx.passthroughBufferedTextualToolCallContent ||
+ containsTextualToolCallCandidate(incomingDelta)
+ ) {
+ const parsedCandidate = parseTextualToolCallCandidate(bufferedCandidate);
+ if (parsedCandidate?.kind === "complete") {
+ const collectedToolCall = collectPassthroughTextualToolCall(
+ bufferedCandidate,
+ ctx.passthroughToolCalls,
+ ctx.allowedToolNames
+ );
+ if (collectedToolCall) {
+ ctx.passthroughHasToolCalls = true;
+ const responseToolCallEvents =
+ buildResponsesFunctionCallEvents(collectedToolCall);
+ output = formatSSEDataEvents(responseToolCallEvents);
+ ctx.clientPayloadCollector.push(...responseToolCallEvents);
+ ctx.reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(ctx.encoder.encode(output));
+ injectedUsage = true;
+ } else {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ ctx.passthroughBufferedTextualToolCallContent = "";
+ parsed.delta = "";
+ } else if (parsedCandidate?.kind === "partial") {
+ ctx.passthroughBufferedTextualToolCallContent = appendBoundedText(
+ ctx.passthroughBufferedTextualToolCallContent,
+ incomingDelta
+ );
+ parsed.delta = "";
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ } else {
+ if (ctx.passthroughBufferedTextualToolCallContent) {
+ parsed.delta =
+ ctx.passthroughBufferedTextualToolCallContent + incomingDelta;
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ ctx.passthroughAccumulatedContent = appendBoundedText(
+ ctx.passthroughAccumulatedContent,
+ ctx.passthroughBufferedTextualToolCallContent + incomingDelta
+ );
+ ctx.passthroughBufferedTextualToolCallContent = "";
+ }
+ } else {
+ ctx.passthroughAccumulatedContent = appendBoundedText(
+ ctx.passthroughAccumulatedContent,
+ incomingDelta
+ );
+ }
+ }
+ if (parsed.type === "response.failed") {
+ failurePayload = normalizeStreamFailurePayload(parsed);
+ }
+ if (
+ parsed.type === "response.reasoning_summary_text.delta" ||
+ parsed.type === "response.reasoning_summary_text.done" ||
+ parsed.type === "response.reasoning_summary_part.done"
+ ) {
+ const reasoningKey = ctx.getResponsesReasoningKey(parsed);
+ if (reasoningKey) {
+ ctx.passthroughResponsesReasoningSummarySeen.add(reasoningKey);
+ }
+ }
+ if (
+ parsed.type === "response.output_item.added" &&
+ parsed.item?.type === "function_call"
+ ) {
+ const item =
+ parsed.item && typeof parsed.item === "object" && !Array.isArray(parsed.item)
+ ? { ...(parsed.item as JsonRecord) }
+ : null;
+ const pendingKey =
+ item && typeof item.id === "string"
+ ? item.id
+ : item && typeof item.call_id === "string"
+ ? item.call_id
+ : null;
+ if (item && pendingKey) {
+ if (typeof item.arguments !== "string") {
+ item.arguments = "";
+ }
+ ctx.passthroughResponsesPendingFunctionCalls.set(pendingKey, item);
+ ctx.passthroughResponsesCurrentFunctionCallKey = pendingKey;
+ }
+ }
+ if (parsed.type === "response.function_call_arguments.delta") {
+ const pendingKey =
+ typeof parsed.item_id === "string"
+ ? parsed.item_id
+ : ctx.passthroughResponsesCurrentFunctionCallKey;
+ const pending = pendingKey
+ ? ctx.passthroughResponsesPendingFunctionCalls.get(pendingKey)
+ : undefined;
+ if (pending && typeof parsed.delta === "string") {
+ const previousArgs =
+ typeof pending.arguments === "string" ? pending.arguments : "";
+ pending.arguments = previousArgs + parsed.delta;
+ }
+ }
+ if (parsed.type === "response.function_call_arguments.done") {
+ const pendingKey =
+ typeof parsed.item_id === "string"
+ ? parsed.item_id
+ : ctx.passthroughResponsesCurrentFunctionCallKey;
+ const pending = pendingKey
+ ? ctx.passthroughResponsesPendingFunctionCalls.get(pendingKey)
+ : undefined;
+ if (pending) {
+ if (typeof parsed.arguments === "string") {
+ pending.arguments = parsed.arguments;
+ }
+ pushUniqueResponsesOutputItems(ctx.passthroughResponsesOutputItems, [
+ pending,
+ ]);
+ }
+ }
+ // Capture each completed output item so the final
+ // response.completed snapshot can be backfilled when upstream
+ // returns an empty `output` (happens with store: false).
+ if (parsed.type === "response.output_item.done" && parsed.item) {
+ const reasoningSummaryInjected =
+ ctx.ensureVisibleResponsesReasoningSummary(parsed);
+ ctx.emitSyntheticResponsesReasoningSummary(controller, parsed);
+ pushUniqueResponsesOutputItems(ctx.passthroughResponsesOutputItems, [
+ parsed.item,
+ ]);
+ if (reasoningSummaryInjected) {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ if (parsed.item?.type === "function_call") {
+ const pendingKey =
+ typeof parsed.item.id === "string"
+ ? parsed.item.id
+ : typeof parsed.item.call_id === "string"
+ ? parsed.item.call_id
+ : null;
+ if (pendingKey) {
+ ctx.passthroughResponsesPendingFunctionCalls.delete(pendingKey);
+ if (ctx.passthroughResponsesCurrentFunctionCallKey === pendingKey) {
+ ctx.passthroughResponsesCurrentFunctionCallKey = null;
+ }
+ }
+ }
+ }
+ if (
+ parsed.type === "response.completed" &&
+ Array.isArray(parsed.response?.output) &&
+ parsed.response.output.length > 0
+ ) {
+ pushUniqueResponsesOutputItems(
+ ctx.passthroughResponsesOutputItems,
+ parsed.response.output
+ );
+ }
+ if (
+ parsed.type === "response.completed" &&
+ ctx.passthroughResponsesPendingFunctionCalls.size > 0
+ ) {
+ pushUniqueResponsesOutputItems(ctx.passthroughResponsesOutputItems, [
+ ...ctx.passthroughResponsesPendingFunctionCalls.values(),
+ ]);
+ ctx.passthroughResponsesPendingFunctionCalls.clear();
+ ctx.passthroughResponsesCurrentFunctionCallKey = null;
+ }
+ // Two transport-level fixes for Responses passthrough:
+ // 1) Strip echoed `instructions` + `tools` from lifecycle
+ // events — they can balloon a single SSE event past
+ // 100 KB and break parsers (e.g. GitHub Copilot CLI).
+ // 2) Backfill `response.completed.response.output` when
+ // upstream sent it empty (store: false) — some clients
+ // build their tool-call list from that snapshot rather
+ // than from per-item events.
+ const textualToolCallBackfilled =
+ parsed.type === "response.completed" && ctx.passthroughToolCalls.size > 0;
+ if (textualToolCallBackfilled) {
+ parsed = toResponsesCompletedWithToolCalls(parsed as JsonRecord, [
+ ...ctx.passthroughToolCalls.values(),
+ ]) as typeof parsed;
+ }
+ const stripped = stripResponsesLifecycleEcho(parsed);
+ const backfilled = backfillResponsesCompletedOutput(
+ parsed,
+ ctx.passthroughResponsesOutputItems
+ );
+ if (
+ stripped ||
+ backfilled ||
+ textualToolCallBackfilled ||
+ responsesIdsNormalized
+ ) {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ } else if (isClaudeSSE) {
+ // Claude SSE: extract ctx.usage, track content, forward as-is
+ const extracted = extractUsage(parsed);
+ if (extracted) {
+ // Non-destructive merge: never overwrite a positive value with 0
+ // message_start carries input_tokens, message_delta carries output_tokens;
+ if (!ctx.usage) ctx.usage = {};
+ const u = ctx.usage;
+ const eu = extracted as UsageTokenRecord;
+ if (eu.prompt_tokens > 0) u.prompt_tokens = eu.prompt_tokens;
+ if (eu.completion_tokens > 0) u.completion_tokens = eu.completion_tokens;
+ if (eu.total_tokens > 0) u.total_tokens = eu.total_tokens;
+ if (eu.cache_read_input_tokens)
+ u.cache_read_input_tokens = eu.cache_read_input_tokens;
+ if (eu.cache_creation_input_tokens)
+ u.cache_creation_input_tokens = eu.cache_creation_input_tokens;
+ }
+ if (
+ shouldInjectClaudeEmptyResponseBeforeCurrentEvent(
+ ctx.claudeEmptyResponseLifecycle,
+ parsed
+ )
+ ) {
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: true,
+ includeMessageDelta:
+ parsed.type === "message_stop" &&
+ !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: false,
+ });
+ }
+ updateClaudeEmptyResponseLifecycle(ctx.claudeEmptyResponseLifecycle, parsed);
+ const restoredToolName = restoreClaudePassthroughToolUseName(
+ parsed,
+ ctx.toolNameMap
+ );
+ // Track content length and accumulate from Claude format
+ if (parsed.delta?.text) {
+ ctx.totalContentLength += parsed.delta.text.length;
+ ctx.passthroughAccumulatedContent = appendBoundedText(
+ ctx.passthroughAccumulatedContent,
+ parsed.delta.text
+ );
+ }
+ if (parsed.delta?.thinking) {
+ ctx.totalContentLength += parsed.delta.thinking.length;
+ ctx.passthroughAccumulatedContent = appendBoundedText(
+ ctx.passthroughAccumulatedContent,
+ parsed.delta.thinking
+ );
+ }
+ if (restoredToolName) {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ } else {
+ // ----------------------------------------------------
+ // READ THIS !!!
+ // ----------------------------------------------------
+ // IF YOU ARE AI OR AI AGENT DONT TOUCH THIS CODE !!!
+ // YOU WILL BREAK THIS CODE !!!
+ // GOOD LUCK
+ // ----------------------------------------------------
+ // Chat Completions: full sanitization pipeline
+
+ // OpenAI-compatible streaming with `stream_options.include_usage=true`
+ // ends with a ctx.usage-only chunk where `choices` is deliberately `[]`.
+ // Forward that standards-compliant chunk instead of turning it into an
+ // empty-response error.
+ //
+ // For a malformed empty `choices: []` chunk WITHOUT valid ctx.usage we DROP
+ // it (log server-side only). We must NOT inject an assistant-content
+ // chunk like "[OmniRoute] Upstream returned an empty response. Please
+ // retry." with finish_reason: "stop" — clients (Goose/opencode) feed that
+ // text back as a turn and spin in a retry loop. This restores the #3400
+ // behavior that #3422 inadvertently reverted (regression #3388/#3502).
+ if (Array.isArray(parsed.choices) && parsed.choices.length === 0) {
+ const emptyChoicesUsage = extractUsage(parsed) ?? parsed.usage;
+ if (hasValidUsage(emptyChoicesUsage)) {
+ ctx.usage = emptyChoicesUsage;
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ clientPayload = parsed;
+ ctx.clientPayloadCollector.push(clientPayload);
+ ctx.reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(ctx.encoder.encode(output));
+ continue;
+ }
+
+ console.warn(
+ `[STREAM] Upstream returned empty choices array (${ctx.provider || "provider"}:${ctx.model || "unknown"}) — dropping chunk`
+ );
+ continue;
+ }
+
+ // Detect reasoning alias before sanitization strips it
+ const hadReasoningAlias = !!(
+ parsed.choices?.[0]?.delta?.reasoning &&
+ typeof parsed.choices[0].delta.reasoning === "string" &&
+ !parsed.choices[0].delta.reasoning_content
+ );
+ const hadNonStringToolCallId = Array.isArray(parsed.choices)
+ ? parsed.choices.some(
+ (choice) =>
+ Array.isArray(choice?.delta?.tool_calls) &&
+ choice.delta.tool_calls.some(
+ (tc) => tc?.id != null && typeof tc.id !== "string"
+ )
+ )
+ : false;
+ const hadNonStringTopLevelId =
+ parsed?.id != null && typeof parsed.id !== "string";
+
+ parsed = sanitizeStreamingChunk(parsed);
+ if (
+ parsed &&
+ typeof parsed === "object" &&
+ !Array.isArray(parsed) &&
+ (parsed as Record)[OMIT_STREAMING_CHUNK_MARKER] === true
+ ) {
+ continue;
+ }
+
+ const idFixed = hadNonStringTopLevelId ? false : fixInvalidId(parsed);
+
+ if (!hasValuableContent(parsed, FORMATS.OPENAI)) {
+ continue;
+ }
+
+ const delta = parsed.choices?.[0]?.delta;
+ let textualToolCallConverted = false;
+ let toolCallIdCoerced = false;
+
+ // Extract tags from streaming content
+ if (delta?.content && typeof delta.content === "string") {
+ const { content, thinking } = extractThinkingFromContent(delta.content);
+ delta.content = content;
+ if (thinking && !delta.reasoning_content) {
+ delta.reasoning_content = thinking;
+ }
+ }
+
+ // Split combined reasoning+content deltas into separate SSE events.
+ // Standard OpenAI streaming never mixes both fields in one delta;
+ // clients (e.g. LobeChat) may skip content when reasoning_content
+ // is present, causing the first content token to be lost.
+ if (delta?.reasoning_content && delta?.content) {
+ const reasoningChunk = JSON.parse(JSON.stringify(parsed));
+ const rDelta = reasoningChunk.choices[0].delta;
+ delete rDelta.content;
+ reasoningChunk.choices[0].finish_reason = null;
+ delete reasoningChunk.usage;
+ const rOutput = `data: ${JSON.stringify(reasoningChunk)}\n`;
+ ctx.passthroughAccumulatedReasoning = appendBoundedText(
+ ctx.passthroughAccumulatedReasoning,
+ delta.reasoning_content
+ );
+ ctx.totalContentLength += delta.reasoning_content.length;
+ ctx.clientPayloadCollector.push(reasoningChunk);
+ ctx.reqLogger?.appendConvertedChunk?.(rOutput);
+ controller.enqueue(ctx.encoder.encode(rOutput));
+ controller.enqueue(ctx.encoder.encode("\n"));
+ delete delta.reasoning_content;
+ }
+
+ // Track whether we need to re-serialize (separate from injectedUsage
+ // to avoid blocking subsequent finish_reason / ctx.usage mutations)
+ const needsReserialization =
+ hadReasoningAlias || (delta?.content === "" && delta?.reasoning_content);
+
+ // T18: Track if we saw tool calls & accumulate for call log
+ if (delta?.tool_calls && delta.tool_calls.length > 0) {
+ ctx.passthroughHasToolCalls = true;
+ ctx.lastToolCallChunkTime = Date.now();
+ for (const tc of delta.tool_calls) {
+ // Note: sanitizeStreamingChunk above already coerces non-string
+ // tool_call IDs, but this defensive check catches edge cases
+ // where sanitize didn't run (e.g. flush path shortcuts).
+ if (tc?.id != null && typeof tc.id !== "string") {
+ tc.id = String(tc.id);
+ toolCallIdCoerced = true;
+ }
+ // Key by index first — id only appears on the first delta in OpenAI streaming
+ let key: string;
+ if (Number.isInteger(tc?.index)) {
+ key = `idx:${tc.index}`;
+ } else if (tc?.id != null) {
+ key = `id:${tc.id}`;
+ } else {
+ key = `seq:${++ctx.passthroughToolCallSeq}`;
+ }
+ const existing = ctx.passthroughToolCalls.get(key);
+ const deltaArgs =
+ typeof tc?.function?.arguments === "string" ? tc.function.arguments : "";
+ if (!existing) {
+ ctx.passthroughToolCalls.set(key, {
+ id: tc?.id != null ? String(tc.id) : null,
+ index: Number.isInteger(tc?.index)
+ ? tc.index
+ : ctx.passthroughToolCalls.size,
+ type: tc?.type || "function",
+ function: {
+ name: tc?.function?.name || "",
+ arguments: deltaArgs,
+ },
+ });
+ } else {
+ if (tc?.id) existing.id = existing.id || String(tc.id);
+ if (tc?.function?.name && !existing.function.name)
+ existing.function.name = tc.function.name;
+ existing.function.arguments += deltaArgs;
+ }
+ }
+ }
+
+ const content = delta?.content || delta?.reasoning_content;
+ if (typeof content === "string") {
+ ctx.totalContentLength += content.length;
+
+ if (!ctx.contentAfterToolSeen) {
+ const toolTs = ctx.toolFinishTime || ctx.pendingToolFinishTime;
+ const lastChunkTs = ctx.lastToolCallChunkTime;
+ if (toolTs || lastChunkTs) {
+ ctx.contentAfterToolSeen = true;
+ const now = Date.now();
+ try {
+ recordToolLatency(
+ ctx.provider || "unknown",
+ toolTs ? now - toolTs : null,
+ lastChunkTs ? now - lastChunkTs : null
+ );
+ } catch {}
+ ctx.pendingToolFinishTime = null;
+ }
+ }
+ }
+ {
+ const guarded = ctx.applyTextualToolCallStreamingGuard(
+ parsed as Record
+ );
+ parsed = guarded.parsed as typeof parsed;
+ textualToolCallConverted = guarded.textualToolCallConverted;
+ }
+ if (typeof delta?.reasoning_content === "string")
+ ctx.passthroughAccumulatedReasoning = appendBoundedText(
+ ctx.passthroughAccumulatedReasoning,
+ delta.reasoning_content
+ );
+
+ const extracted = extractUsage(parsed);
+ if (extracted) {
+ ctx.usage = extracted;
+ }
+
+ const isFinishChunk = parsed.choices?.[0]?.finish_reason;
+
+ if (isFinishChunk && ctx.passthroughHasToolCalls) {
+ ctx.toolFinishTime = Date.now();
+ try {
+ markToolFinish(ctx.sessionId);
+ } catch {}
+ }
+
+ // T18: Normalize finish_reason to 'tool_calls' if tool calls were used
+ if (
+ isFinishChunk &&
+ ctx.passthroughHasToolCalls &&
+ parsed.choices[0].finish_reason !== "tool_calls"
+ ) {
+ parsed.choices[0].finish_reason = "tool_calls";
+ // If we modify it, we must output the modified object
+ if (!injectedUsage && hasValidUsage(parsed.usage)) {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ }
+ if (
+ isFinishChunk &&
+ !hasValidUsage(parsed.usage) &&
+ !ctx.expectsOpenAIUsageOnlyChunk
+ ) {
+ const estimated = estimateUsage(
+ ctx.body,
+ ctx.totalContentLength,
+ FORMATS.OPENAI
+ );
+ parsed.usage = filterUsageForFormat(estimated, FORMATS.OPENAI);
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ ctx.usage = estimated;
+ injectedUsage = true;
+ } else if (isFinishChunk && ctx.usage) {
+ const buffered = addBufferToUsage(ctx.usage);
+ parsed.usage = filterUsageForFormat(buffered, FORMATS.OPENAI);
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ } else if (textualToolCallConverted) {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ } else if (
+ idFixed ||
+ needsReserialization ||
+ toolCallIdCoerced ||
+ hadNonStringToolCallId ||
+ hadNonStringTopLevelId
+ ) {
+ output = `data: ${JSON.stringify(parsed)}\n`;
+ injectedUsage = true;
+ }
+ }
+
+ clientPayload = parsed;
+ } catch {}
+ }
+
+ if (!injectedUsage) {
+ if (line.startsWith("data:") && !line.startsWith("data: ")) {
+ output = "data: " + line.slice(5) + "\n";
+ } else {
+ output = line + "\n";
+ }
+ }
+
+ if (
+ !trimmed &&
+ ctx.pendingPassthroughEventLine &&
+ !ctx.pendingPassthroughEventEmitted
+ ) {
+ output = `${ctx.pendingPassthroughEventLine}\n${output}`;
+ ctx.pendingPassthroughEventEmitted = true;
+ }
+
+ output = ctx.maybePrefixPendingPassthroughEvent(output, line);
+
+ if (clientPayload) {
+ ctx.clientPayloadCollector.push(clientPayload);
+ }
+
+ ctx.reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(ctx.encoder.encode(output));
+ if (failurePayload) {
+ if (ctx.onFailure) {
+ try {
+ void ctx.onFailure(failurePayload);
+ } catch {}
+ }
+ ctx.clearIdleTimer();
+ trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false);
+ controller.error(
+ markPendingRequestCleared(new Error(failurePayload.message || "Upstream failure"))
+ );
+ return;
+ }
+ if (!trimmed) {
+ ctx.clearPendingPassthroughEvent();
+ }
+ continue;
+ }
+
+ // Translate ctx.mode
+ if (!trimmed) continue;
+
+ if (ctx.state?.upstreamError) {
+ continue;
+ }
+
+ const parsed = parseSSELine(trimmed);
+ if (!parsed) continue;
+ ctx.providerPayloadCollector.push(parsed);
+
+ if (parsed && parsed.done) {
+ continue;
+ }
+
+ if (parsed.choices?.[0]?.delta?.tool_calls) {
+ ctx.lastToolCallChunkTime = Date.now();
+ }
+ if (parsed.choices?.[0]?.finish_reason === "tool_calls") {
+ ctx.toolFinishTime = Date.now();
+ try {
+ markToolFinish(ctx.sessionId);
+ } catch {}
+ }
+
+ // Track content length and accumulate for call log (from raw ctx.provider chunk, so content is never missed)
+ // Do this before translation so we capture content regardless of translator output shape
+
+ // Claude format
+ if (parsed.delta?.text) {
+ const t = parsed.delta.text;
+ ctx.totalContentLength += t.length;
+ if (ctx.state?.accumulatedContent !== undefined && typeof t === "string")
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, t);
+ }
+ if (parsed.delta?.thinking) {
+ const t = parsed.delta.thinking;
+ ctx.totalContentLength += t.length;
+ if (ctx.state?.accumulatedContent !== undefined && typeof t === "string")
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, t);
+ }
+
+ // OpenAI format
+ if (parsed.choices?.[0]?.delta?.content) {
+ const c = parsed.choices[0].delta.content;
+ if (typeof c === "string") {
+ ctx.totalContentLength += c.length;
+ if (ctx.state?.accumulatedContent !== undefined)
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, c);
+ } else if (Array.isArray(c)) {
+ for (const part of c) {
+ if (part?.text && typeof part.text === "string") {
+ ctx.totalContentLength += part.text.length;
+ if (ctx.state?.accumulatedContent !== undefined)
+ ctx.state.accumulatedContent = appendBoundedText(
+ ctx.state.accumulatedContent,
+ part.text
+ );
+ }
+ }
+ }
+ }
+ if (parsed.choices?.[0]?.delta?.reasoning_content) {
+ const r = parsed.choices[0].delta.reasoning_content;
+ if (typeof r === "string") {
+ ctx.totalContentLength += r.length;
+ if (ctx.state?.accumulatedContent !== undefined)
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, r);
+ }
+ }
+ // Normalize `reasoning` alias → `reasoning_content` (NVIDIA kimi-k2.5 etc.)
+ if (
+ parsed.choices?.[0]?.delta?.reasoning &&
+ !parsed.choices?.[0]?.delta?.reasoning_content
+ ) {
+ const r = parsed.choices[0].delta.reasoning;
+ if (typeof r === "string") {
+ parsed.choices[0].delta.reasoning_content = r;
+ delete parsed.choices[0].delta.reasoning;
+ ctx.totalContentLength += r.length;
+ if (ctx.state?.accumulatedContent !== undefined)
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, r);
+ }
+ }
+
+ // Gemini / Cloud Code format - may have multiple parts
+ // Cloud Code API wraps in { response: { candidates: [...] } }, so unwrap.
+ // Only applies to Gemini-family formats — skip for OpenAI, Claude, etc.
+ const isGeminiFormat =
+ ctx.targetFormat === FORMATS.GEMINI ||
+ ctx.targetFormat === FORMATS.GEMINI_CLI ||
+ ctx.targetFormat === FORMATS.ANTIGRAVITY;
+ const geminiChunk = isGeminiFormat ? unwrapGeminiChunk(parsed) : parsed;
+ if (geminiChunk.candidates?.[0]?.content?.parts) {
+ for (const part of geminiChunk.candidates[0].content.parts) {
+ if (part.text && typeof part.text === "string") {
+ ctx.totalContentLength += part.text.length;
+ if (ctx.state?.accumulatedContent !== undefined)
+ ctx.state.accumulatedContent = appendBoundedText(
+ ctx.state.accumulatedContent,
+ part.text
+ );
+ }
+ }
+ }
+
+ // Generic fallback: delta string, top-level content/text (e.g. some SSE payloads)
+ if (ctx.state?.accumulatedContent !== undefined) {
+ if (typeof (parsed as JsonRecord).delta === "string") {
+ const d = (parsed as JsonRecord).delta as string;
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, d);
+ ctx.totalContentLength += d.length;
+ }
+ if (typeof (parsed as JsonRecord).content === "string") {
+ const c = (parsed as JsonRecord).content as string;
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, c);
+ ctx.totalContentLength += c.length;
+ }
+ if (typeof (parsed as JsonRecord).text === "string") {
+ const t = (parsed as JsonRecord).text as string;
+ ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, t);
+ ctx.totalContentLength += t.length;
+ }
+ }
+
+ const translateHasContent =
+ typeof parsed.delta?.text === "string" ||
+ typeof parsed.choices?.[0]?.delta?.content === "string" ||
+ typeof parsed.choices?.[0]?.delta?.reasoning_content === "string";
+ if (translateHasContent && !ctx.contentAfterToolSeen) {
+ const toolTs = ctx.toolFinishTime || ctx.pendingToolFinishTime;
+ const lastChunkTs = ctx.lastToolCallChunkTime;
+ if (toolTs || lastChunkTs) {
+ ctx.contentAfterToolSeen = true;
+ const now = Date.now();
+ try {
+ recordToolLatency(
+ ctx.provider || "unknown",
+ toolTs ? now - toolTs : null,
+ lastChunkTs ? now - lastChunkTs : null
+ );
+ } catch {}
+ ctx.pendingToolFinishTime = null;
+ }
+ }
+
+ // Extract ctx.usage
+ const extracted = extractUsage(parsed);
+ if (extracted) ctx.state.usage = extracted; // Keep original ctx.usage for logging
+
+ // Translate: ctx.targetFormat -> openai -> ctx.sourceFormat
+ const translated = translateResponse(
+ ctx.targetFormat,
+ ctx.sourceFormat,
+ parsed,
+ ctx.state
+ );
+
+ // Log OpenAI intermediate chunks (if available)
+ for (const item of getOpenAIIntermediateChunks(translated)) {
+ const openaiOutput = formatSSE(item, FORMATS.OPENAI);
+ ctx.reqLogger?.appendOpenAIChunk?.(openaiOutput);
+ }
+
+ if (translated?.length > 0) {
+ for (const item of translated) {
+ ctx.emitTranslatedClientItem(controller, item);
+ }
+ }
+ }
+ },
+
+ async flush(controller) {
+ // Clean up idle watchdog timer
+ if (ctx.idleTimer) {
+ ctx.clearIdleTimer();
+ }
+ if (ctx.streamTimedOut) {
+ return;
+ }
+ trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false);
+ try {
+ const remaining = ctx.decoder.decode();
+ if (remaining) ctx.buffer += remaining;
+
+ if (ctx.mode === STREAM_MODE.PASSTHROUGH) {
+ const bufferedLine = ctx.buffer.trim();
+ if (ctx.skipPassthroughEvent || /^event:\s*keepalive\b/i.test(bufferedLine)) {
+ ctx.skipPassthroughEvent = false;
+ ctx.clearPendingPassthroughEvent();
+ } else if (ctx.buffer) {
+ let output = ctx.buffer;
+ if (ctx.buffer.startsWith("data:") && !ctx.buffer.startsWith("data: ")) {
+ output = "data: " + ctx.buffer.slice(5);
+ }
+ const bufferedPayload = parseSSELine(bufferedLine);
+ if (bufferedPayload) {
+ ctx.providerPayloadCollector.push(bufferedPayload);
+ if (
+ shouldInjectClaudeEmptyResponseBeforeCurrentEvent(
+ ctx.claudeEmptyResponseLifecycle,
+ bufferedPayload
+ )
+ ) {
+ const eventType = getClaudeEventType(bufferedPayload);
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: true,
+ includeMessageDelta:
+ eventType === "message_stop" &&
+ !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: false,
+ });
+ }
+ if (isClaudeEventPayload(bufferedPayload)) {
+ updateClaudeEmptyResponseLifecycle(
+ ctx.claudeEmptyResponseLifecycle,
+ bufferedPayload
+ );
+ }
+ ctx.clientPayloadCollector.push(bufferedPayload);
+
+ // Normalize numeric IDs for final buffered data: chunk (same as transform path)
+ if (typeof bufferedPayload === "object" && !Array.isArray(bufferedPayload)) {
+ const flushedParsed = bufferedPayload as JsonRecord;
+ const flushedType =
+ typeof flushedParsed.type === "string" ? flushedParsed.type : "";
+ const isResponses = flushedType.startsWith("response.");
+ const isClaude = isClaudeEventPayload(flushedParsed);
+ if (isResponses) {
+ if (normalizeResponsesSseIds(flushedParsed)) {
+ output = `data: ${JSON.stringify(flushedParsed)}\n`;
+ }
+ } else if (!isClaude) {
+ let flushChanged = false;
+ const flushedHadNonStringTopLevelId =
+ flushedParsed?.id != null && typeof flushedParsed.id !== "string";
+ if (flushedHadNonStringTopLevelId) {
+ flushedParsed.id = String(flushedParsed.id);
+ flushChanged = true;
+ }
+ if (Array.isArray(flushedParsed.choices)) {
+ for (const choice of flushedParsed.choices as JsonRecord[]) {
+ const tcs = (choice as JsonRecord | undefined)?.delta as
+ | JsonRecord
+ | undefined;
+ if (Array.isArray(tcs?.tool_calls)) {
+ for (const tc of tcs.tool_calls as JsonRecord[]) {
+ if (tc?.id != null && typeof tc.id !== "string") {
+ tc.id = String(tc.id);
+ flushChanged = true;
+ }
+ }
+ }
+ }
+ }
+ if (flushChanged) {
+ output = `data: ${JSON.stringify(flushedParsed)}\n`;
+ }
+ }
+ }
+ }
+ if (
+ !bufferedLine &&
+ ctx.pendingPassthroughEventLine &&
+ !ctx.pendingPassthroughEventEmitted
+ ) {
+ output = `${ctx.pendingPassthroughEventLine}\n${output}`;
+ ctx.pendingPassthroughEventEmitted = true;
+ }
+ output = ctx.maybePrefixPendingPassthroughEvent(output, ctx.buffer);
+ ctx.reqLogger?.appendConvertedChunk?.(output);
+ controller.enqueue(ctx.encoder.encode(output));
+ }
+
+ if (shouldInjectClaudeEmptyResponseOnFlush(ctx.claudeEmptyResponseLifecycle)) {
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: true,
+ includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop,
+ });
+ } else if (
+ shouldInjectClaudeMissingFinalizersOnFlush(ctx.claudeEmptyResponseLifecycle)
+ ) {
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: false,
+ includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop,
+ });
+ }
+ ctx.clearPendingPassthroughEvent();
+
+ if (ctx.passthroughBufferedTextualToolCallContent) {
+ // Flush any remaining buffered content as plain text.
+ // Previously gated on !includes("Arguments:"), which silently dropped
+ // incomplete tool-call headers (ctx.buffer held "Arguments:" but JSON was
+ // never finished before stream ended) — fix #3355 bug 2.
+ let flushOutput = "";
+ if (ctx.clientExpectsResponsesStream) {
+ const syntheticChunk = {
+ type: "response.output_text.delta",
+ delta: ctx.passthroughBufferedTextualToolCallContent,
+ };
+ flushOutput = `data: ${JSON.stringify(syntheticChunk)}\n\n`;
+ } else if (ctx.clientExpectsClaudeStream) {
+ const syntheticChunk = {
+ type: "content_block_delta",
+ index: 0,
+ delta: {
+ type: "text_delta",
+ text: ctx.passthroughBufferedTextualToolCallContent,
+ },
+ };
+ flushOutput = `data: ${JSON.stringify(syntheticChunk)}\n\n`;
+ } else {
+ const syntheticChunk = {
+ id: ctx.passthroughResponsesId || `chatcmpl-${Date.now()}`,
+ object: "chat.completion.chunk",
+ created: Math.floor(Date.now() / 1000),
+ model: ctx.model || "unknown",
+ choices: [
+ {
+ index: 0,
+ delta: {
+ content: ctx.passthroughBufferedTextualToolCallContent,
+ },
+ finish_reason: null,
+ },
+ ],
+ };
+ flushOutput = `data: ${JSON.stringify(syntheticChunk)}\n\n`;
+ }
+ ctx.reqLogger?.appendConvertedChunk?.(flushOutput);
+ controller.enqueue(ctx.encoder.encode(flushOutput));
+ ctx.passthroughAccumulatedContent = appendBoundedText(
+ ctx.passthroughAccumulatedContent,
+ ctx.passthroughBufferedTextualToolCallContent
+ );
+ ctx.passthroughBufferedTextualToolCallContent = "";
+ }
+
+ // Estimate ctx.usage if ctx.provider didn't return valid ctx.usage
+ if (!hasValidUsage(ctx.usage) && ctx.totalContentLength > 0) {
+ ctx.usage = estimateUsage(
+ ctx.body,
+ ctx.totalContentLength,
+ ctx.sourceFormat || FORMATS.OPENAI
+ );
+ }
+
+ if (hasValidUsage(ctx.usage)) {
+ logUsage(ctx.provider, ctx.usage, ctx.model, ctx.connectionId, ctx.apiKeyInfo);
+ } else {
+ appendRequestLog({
+ model: ctx.model,
+ provider: ctx.provider,
+ connectionId: ctx.connectionId,
+ tokens: null,
+ status: "200 OK",
+ }).catch(() => {});
+ }
+ if (!ctx.doneSent) {
+ await ctx.emitFinalSseMetadata(controller, ctx.usage);
+ ctx.doneSent = true;
+ if (ctx.shouldEmitDoneTerminator) {
+ ctx.clientPayloadCollector.push({ done: true });
+ const doneOutput = "data: [DONE]\n\n";
+ ctx.reqLogger?.appendConvertedChunk?.(doneOutput);
+ controller.enqueue(ctx.encoder.encode(doneOutput));
+ }
+ }
+ // Notify caller for call log persistence (include full response ctx.body with accumulated content)
+ if (ctx.onComplete) {
+ try {
+ const u = ctx.usage as Record | null;
+ const prompt = Number(u?.prompt_tokens ?? u?.input_tokens ?? 0);
+ const completion = Number(u?.completion_tokens ?? u?.output_tokens ?? 0);
+ let content = ctx.passthroughAccumulatedContent.trim() || "";
+ const finalBufferedTextualToolCall =
+ ctx.passthroughBufferedTextualToolCallContent.trim();
+ if (finalBufferedTextualToolCall) {
+ if (
+ collectPassthroughTextualToolCall(
+ finalBufferedTextualToolCall,
+ ctx.passthroughToolCalls,
+ ctx.allowedToolNames
+ )
+ ) {
+ ctx.passthroughHasToolCalls = true;
+ }
+ ctx.passthroughBufferedTextualToolCallContent = "";
+ }
+ if (
+ content &&
+ collectPassthroughTextualToolCall(
+ content,
+ ctx.passthroughToolCalls,
+ ctx.allowedToolNames
+ )
+ ) {
+ ctx.passthroughHasToolCalls = true;
+ content = "";
+ } else if (containsMalformedTextualToolCall(content, ctx.allowedToolNames)) {
+ content = "";
+ }
+ const message: Record = {
+ role: "assistant",
+ content: content || null,
+ };
+ const reasoning = ctx.passthroughAccumulatedReasoning.trim();
+ if (reasoning) {
+ message.reasoning_content = reasoning;
+ }
+ if (ctx.passthroughToolCalls.size > 0) {
+ message.tool_calls = [...ctx.passthroughToolCalls.values()].sort(
+ (a, b) => a.index - b.index
+ );
+ }
+ // Hardening: log empty assistant response after tool completion
+ // for observability — helps diagnose Copilot "Sorry, no response was returned"
+ if (ctx.passthroughHasToolCalls && !content.trim() && !reasoning.trim()) {
+ console.warn(
+ `[STREAM] Empty assistant response after tool_calls completion (${ctx.provider || "provider"}:${ctx.model || "unknown"}) — ctx.sessionId=${ctx.sessionId}`
+ );
+ }
+
+ const responseBody = {
+ choices: [
+ {
+ message,
+ finish_reason: ctx.passthroughHasToolCalls ? "tool_calls" : "stop",
+ },
+ ],
+ usage: {
+ prompt_tokens: prompt,
+ completion_tokens: completion,
+ total_tokens: prompt + completion,
+ },
+ _streamed: true,
+ };
+ ctx.onComplete({
+ status: 200,
+ usage: ctx.usage,
+ responseBody,
+ providerPayload: ctx.providerPayloadCollector.build(
+ buildStreamSummaryFromEvents(
+ ctx.providerPayloadCollector.getEvents(),
+ ctx.sourceFormat,
+ ctx.model
+ ),
+ { includeEvents: false }
+ ),
+ clientPayload: ctx.clientPayloadCollector.build(responseBody, {
+ includeEvents: false,
+ }),
+ });
+ } catch {}
+ }
+ return;
+ }
+
+ // Translate ctx.mode: process remaining ctx.buffer
+ if (ctx.buffer.trim()) {
+ const parsed = parseSSELine(ctx.buffer.trim());
+ if (parsed && !parsed.done) {
+ ctx.providerPayloadCollector.push(parsed);
+ // Extract ctx.usage from remaining ctx.buffer — if the ctx.usage-bearing event
+ // (e.g. response.completed) is the last SSE line, it ends up here
+ // in the flush handler where extractUsage was not called.
+ // Non-destructive merge: some providers send ctx.usage across multiple
+ // events (e.g. prompt_tokens in message_start, completion_tokens
+ // in message_delta). Direct assignment would lose earlier data.
+ const extracted = extractUsage(parsed);
+ if (extracted) {
+ if (!ctx.state.usage) {
+ ctx.state.usage = extracted;
+ } else {
+ const su = ctx.state.usage as Record;
+ const eu = extracted as Record;
+ if (eu.prompt_tokens > 0) su.prompt_tokens = eu.prompt_tokens;
+ if (eu.completion_tokens > 0) su.completion_tokens = eu.completion_tokens;
+ if (eu.total_tokens > 0) su.total_tokens = eu.total_tokens;
+ if (eu.cache_read_input_tokens > 0)
+ su.cache_read_input_tokens = eu.cache_read_input_tokens;
+ if (eu.cache_creation_input_tokens > 0)
+ su.cache_creation_input_tokens = eu.cache_creation_input_tokens;
+ if (eu.cached_tokens > 0) su.cached_tokens = eu.cached_tokens;
+ if (eu.reasoning_tokens > 0) su.reasoning_tokens = eu.reasoning_tokens;
+ }
+ }
+
+ const translated = translateResponse(
+ ctx.targetFormat,
+ ctx.sourceFormat,
+ parsed,
+ ctx.state
+ );
+
+ // Log OpenAI intermediate chunks
+ for (const item of getOpenAIIntermediateChunks(translated)) {
+ const openaiOutput = formatSSE(item, FORMATS.OPENAI);
+ ctx.reqLogger?.appendOpenAIChunk?.(openaiOutput);
+ }
+
+ if (translated?.length > 0) {
+ for (const item of translated) {
+ ctx.emitTranslatedClientItem(controller, item);
+ }
+ }
+ }
+ }
+
+ if (ctx.state?.upstreamError) {
+ const err = ctx.state.upstreamError;
+ trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false);
+ if (ctx.onFailure) {
+ try {
+ void ctx.onFailure({
+ status: err.status,
+ message: err.message,
+ code: err.code,
+ type: err.type,
+ });
+ } catch {}
+ }
+
+ const errorBody = buildErrorBody(err.status, err.message);
+ if (ctx.onComplete) {
+ try {
+ ctx.onComplete({
+ status: err.status,
+ usage: ctx.state?.usage,
+ responseBody: errorBody,
+ providerPayload: ctx.providerPayloadCollector.build(
+ buildStreamSummaryFromEvents(
+ ctx.providerPayloadCollector.getEvents(),
+ ctx.targetFormat,
+ ctx.model
+ ),
+ { includeEvents: false }
+ ),
+ clientPayload: ctx.clientPayloadCollector.build(errorBody, {
+ includeEvents: false,
+ }),
+ });
+ } catch {}
+ }
+
+ ctx.clearIdleTimer();
+ controller.error(
+ markPendingRequestCleared(new Error(err.message || "Upstream failure"))
+ );
+ return;
+ }
+
+ // Flush remaining events (only once at stream end)
+ const flushed = translateResponse(ctx.targetFormat, ctx.sourceFormat, null, ctx.state);
+
+ // Log OpenAI intermediate chunks for flushed events
+ for (const item of getOpenAIIntermediateChunks(flushed)) {
+ const openaiOutput = formatSSE(item, FORMATS.OPENAI);
+ ctx.reqLogger?.appendOpenAIChunk?.(openaiOutput);
+ }
+
+ if (flushed?.length > 0) {
+ for (const item of flushed) {
+ ctx.emitTranslatedClientItem(controller, item);
+ }
+ }
+
+ if (ctx.sourceFormat === FORMATS.CLAUDE) {
+ if (shouldInjectClaudeEmptyResponseOnFlush(ctx.claudeEmptyResponseLifecycle)) {
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: true,
+ includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop,
+ });
+ } else if (
+ shouldInjectClaudeMissingFinalizersOnFlush(ctx.claudeEmptyResponseLifecycle)
+ ) {
+ ctx.emitSyntheticClaudeEmptyResponse(controller, {
+ includeContentBlock: false,
+ includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta,
+ includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop,
+ });
+ }
+ }
+
+ /**
+ * Usage injection strategy:
+ * Usage data (input/output tokens) is injected into the last content chunk
+ * or the finish_reason chunk rather than sent as a separate SSE event.
+ * This ensures all major clients (Claude CLI, Continue, Cursor) receive
+ * ctx.usage data even if they stop reading after the finish signal.
+ * The ctx.usage ctx.buffer (state.usage) accumulates across chunks and is only
+ * emitted once at stream end when merged into the final translated chunk.
+ */
+
+ // Send [DONE] (only if not already sent during transform)
+ if (!ctx.doneSent) {
+ await ctx.emitFinalSseMetadata(
+ controller,
+ ctx.state?.usage as Record | null
+ );
+ ctx.doneSent = true;
+ if (ctx.shouldEmitDoneTerminator) {
+ ctx.clientPayloadCollector.push({ done: true });
+ const doneOutput = "data: [DONE]\n\n";
+ ctx.reqLogger?.appendConvertedChunk?.(doneOutput);
+ controller.enqueue(ctx.encoder.encode(doneOutput));
+ }
+ }
+
+ // Estimate ctx.usage if ctx.provider didn't return valid ctx.usage (for translate ctx.mode)
+ if (!hasValidUsage(ctx.state?.usage) && ctx.totalContentLength > 0) {
+ ctx.state.usage = estimateUsage(ctx.body, ctx.totalContentLength, ctx.sourceFormat);
+ }
+
+ if (hasValidUsage(ctx.state?.usage)) {
+ logUsage(
+ ctx.state?.provider || ctx.targetFormat,
+ ctx.state.usage,
+ ctx.model,
+ ctx.connectionId,
+ ctx.apiKeyInfo
+ );
+ } else {
+ appendRequestLog({
+ model: ctx.model,
+ provider: ctx.provider,
+ connectionId: ctx.connectionId,
+ tokens: null,
+ status: "200 OK",
+ }).catch(() => {});
+ }
+ // Notify caller for call log persistence (include full response ctx.body with accumulated content)
+ if (ctx.onComplete) {
+ try {
+ const u = ctx.state?.usage as Record | null | undefined;
+ const prompt = Number(u?.prompt_tokens ?? u?.input_tokens ?? 0);
+ const completion = Number(u?.completion_tokens ?? u?.output_tokens ?? 0);
+ let content = (ctx.state?.accumulatedContent ?? "").trim() || "";
+ const normalizedToolCalls: ToolCall[] = ctx.state?.toolCalls?.size
+ ? [...ctx.state.toolCalls.values()]
+ .map(
+ (tc: Record): ToolCall => ({
+ id: tc.id != null ? String(tc.id) : null,
+ index: (tc.index as number) ?? (tc.blockIndex as number) ?? 0,
+ type: (tc.type as string) ?? "function",
+ function: (tc.function as ToolCall["function"]) ?? {
+ name: (tc.name as string) ?? "",
+ arguments: "",
+ },
+ })
+ )
+ .sort((a, b) => a.index - b.index)
+ : [];
+ const textualToolCall = parseTextualToolCallFromContent(content);
+ if (textualToolCall) {
+ normalizedToolCalls.push({
+ id: `call_${Date.now()}_${normalizedToolCalls.length}`,
+ index: normalizedToolCalls.length,
+ type: "function",
+ function: {
+ name: textualToolCall.name,
+ arguments: JSON.stringify(textualToolCall.args || {}),
+ },
+ });
+ content = "";
+ } else if (containsMalformedTextualToolCall(content, ctx.allowedToolNames)) {
+ content = "";
+ }
+ const message: Record = {
+ role: "assistant",
+ content: content || null,
+ };
+ const hasToolCalls = normalizedToolCalls.length > 0;
+ if (hasToolCalls) {
+ message.tool_calls = normalizedToolCalls;
+ }
+ const responseBody = {
+ choices: [
+ {
+ message,
+ finish_reason: hasToolCalls ? "tool_calls" : "stop",
+ },
+ ],
+ usage: {
+ prompt_tokens: prompt,
+ completion_tokens: completion,
+ total_tokens: prompt + completion,
+ },
+ _streamed: true,
+ };
+ ctx.onComplete({
+ status: 200,
+ usage: ctx.state?.usage,
+ responseBody,
+ providerPayload: ctx.providerPayloadCollector.build(
+ buildStreamSummaryFromEvents(
+ ctx.providerPayloadCollector.getEvents(),
+ ctx.targetFormat,
+ ctx.model
+ ),
+ { includeEvents: false }
+ ),
+ clientPayload: ctx.clientPayloadCollector.build(responseBody, {
+ includeEvents: false,
+ }),
+ });
+ } catch {}
+ }
+ } catch (error) {
+ console.log(
+ `[STREAM] Error in flush (${ctx.model || "unknown"}):`,
+ error.message || error
+ );
+ }
+ },
+ cancel(reason) {
+ ctx.clearIdleTimer();
+ },
+ },
+ { highWaterMark: 16384 },
+ { highWaterMark: 16384 }
+ );
+}
+
+// Convenience functions for backward compatibility
+export function createSSETransformStreamWithLogger(
+ targetFormat: string,
+ sourceFormat: string,
+ provider: string | null = null,
+ reqLogger: StreamLogger | null = null,
+ toolNameMap: unknown = null,
+ model: string | null = null,
+ connectionId: string | null = null,
+ body: unknown = null,
+ onComplete: ((payload: StreamCompletePayload) => void) | null = null,
+ apiKeyInfo: unknown = null,
+ onFailure: ((payload: StreamFailurePayload) => void | Promise) | null = null,
+ copilotCompatibleReasoning = false
+) {
+ return createSSEStream({
+ mode: STREAM_MODE.TRANSLATE,
+ targetFormat,
+ sourceFormat,
+ provider,
+ reqLogger,
+ toolNameMap,
+ model,
+ connectionId,
+ apiKeyInfo,
+ body,
+ onComplete,
+ onFailure,
+ copilotCompatibleReasoning,
+ });
+}
+
+export function createPassthroughStreamWithLogger(
+ provider: string | null = null,
+ reqLogger: StreamLogger | null = null,
+ toolNameMap: unknown = null,
+ model: string | null = null,
+ connectionId: string | null = null,
+ body: unknown = null,
+ onComplete: ((payload: StreamCompletePayload) => void) | null = null,
+ apiKeyInfo: unknown = null,
+ onFailure: ((payload: StreamFailurePayload) => void | Promise) | null = null,
+ clientResponseFormat: string | null = null
+) {
+ return createSSEStream({
+ mode: STREAM_MODE.PASSTHROUGH,
+ provider,
+ reqLogger,
+ toolNameMap,
+ model,
+ connectionId,
+ apiKeyInfo,
+ body,
+ onComplete,
+ onFailure,
+ clientResponseFormat,
+ });
+}
diff --git a/open-sse/utils/stream/textualToolCalls.ts b/open-sse/utils/stream/textualToolCalls.ts
new file mode 100644
index 00000000000..275cc7e8563
--- /dev/null
+++ b/open-sse/utils/stream/textualToolCalls.ts
@@ -0,0 +1,85 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { asRecord } from "./utils.ts";
+import { JsonRecord, ToolCall } from "./types.ts";
+
+export function parseTextualToolCallFromContent(text: unknown): { name: string; args: unknown } | null {
+ const candidate = parseTextualToolCallCandidate(text);
+ return candidate?.kind === "complete" ? { name: candidate.name, args: candidate.args } : null;
+}
+
+export function containsTextualToolCallCandidate(text: unknown): boolean {
+ return parseTextualToolCallCandidate(text) !== null;
+}
+
+export function containsMalformedTextualToolCall(
+ text: unknown,
+ allowedToolNames?: Set | null
+): boolean {
+ if (typeof text !== "string") return false;
+ const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, "");
+
+ let searchIdx = 0;
+ while (true) {
+ const idx = normalized.indexOf("[Tool call:", searchIdx);
+ if (idx === -1) break;
+
+ const candidate = normalized.slice(idx);
+ if (isValidToolCallHeaderPrefix(candidate)) {
+ const parsed = parseTextualToolCallFromContent(candidate);
+ if (parsed) {
+ if (allowedToolNames?.size && !allowedToolNames.has(parsed.name)) {
+ return true;
+ }
+ } else {
+ return true;
+ }
+ }
+
+ searchIdx = idx + 1;
+ }
+ return false;
+}
+
+export function extractAllowedToolNames(body: unknown): Set | null {
+ const record = asRecord(body);
+ const tools = record.tools;
+ if (!Array.isArray(tools)) return null;
+ const names = new Set();
+ for (const tool of tools) {
+ if (!tool || typeof tool !== "object" || Array.isArray(tool)) continue;
+ const item = tool as JsonRecord;
+ const directName = typeof item.name === "string" ? item.name.trim() : "";
+ const fn =
+ item.function && typeof item.function === "object" && !Array.isArray(item.function)
+ ? (item.function as JsonRecord)
+ : null;
+ const functionName = typeof fn?.name === "string" ? fn.name.trim() : "";
+ const name = functionName || directName;
+ if (name) names.add(name);
+ }
+ return names.size > 0 ? names : null;
+}
+
+export function collectPassthroughTextualToolCall(
+ text: string,
+ toolCalls: Map,
+ allowedToolNames?: Set | null
+): ToolCall | null {
+ const parsed = parseTextualToolCallFromContent(text);
+ if (!parsed) return null;
+ if (allowedToolNames?.size && !allowedToolNames.has(parsed.name)) return null;
+ const key = `textual:${toolCalls.size}`;
+ const toolCall: ToolCall = {
+ id: `call_${Date.now()}_${toolCalls.size}`,
+ index: toolCalls.size,
+ type: "function",
+ function: {
+ name: parsed.name,
+ arguments: JSON.stringify(parsed.args || {}),
+ },
+ };
+ toolCalls.set(key, toolCall);
+ return toolCall;
+}
\ No newline at end of file
diff --git a/open-sse/utils/stream/types.ts b/open-sse/utils/stream/types.ts
new file mode 100644
index 00000000000..728b303532e
--- /dev/null
+++ b/open-sse/utils/stream/types.ts
@@ -0,0 +1,138 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+export type JsonRecord = Record;
+
+export type StreamLogger = {
+ appendProviderChunk?: (value: string) => void;
+ appendConvertedChunk?: (value: string) => void;
+ appendOpenAIChunk?: (value: string) => void;
+};
+
+export type StreamCompletePayload = {
+ status: number;
+ usage: unknown;
+ /** Minimal response body for call log (streaming: usage + note; non-streaming not used) */
+ responseBody?: unknown;
+ providerPayload?: unknown;
+ clientPayload?: unknown;
+};
+
+export type StreamFailurePayload = {
+ status: number;
+ message: string;
+ code?: string;
+ type?: string;
+};
+
+export type StreamOptions = {
+ mode?: string;
+ targetFormat?: string;
+ sourceFormat?: string;
+ clientResponseFormat?: string | null;
+ copilotCompatibleReasoning?: boolean;
+ provider?: string | null;
+ reqLogger?: StreamLogger | null;
+ toolNameMap?: unknown;
+ model?: string | null;
+ connectionId?: string | null;
+ apiKeyInfo?: unknown;
+ body?: unknown;
+ onComplete?: ((payload: StreamCompletePayload) => void) | null;
+ onFailure?: ((payload: StreamFailurePayload) => void | Promise) | null;
+};
+
+export type TranslateState = ReturnType & {
+ provider?: string | null;
+ toolNameMap?: unknown;
+ signatureNamespace?: string | null;
+ usage?: unknown;
+ finishReason?: unknown;
+ copilotCompatibleReasoning?: boolean;
+ /** Accumulated message content for call log response body */
+ accumulatedContent?: string;
+ upstreamError?: {
+ status: number;
+ type: string;
+ code: string;
+ message: string;
+ } | null;
+};
+
+export type ToolCall = {
+ id: string | null;
+ index: number;
+ type: string;
+ function: { name: string; arguments: string };
+};
+
+export type UsageTokenRecord = Record;
+
+export type SSEStreamContext = {
+ mode: string;
+ targetFormat?: string;
+ sourceFormat?: string;
+ clientResponseFormat: string | null;
+ copilotCompatibleReasoning: boolean;
+ provider: string | null;
+ reqLogger: StreamLogger | null;
+ toolNameMap: unknown;
+ model: string | null;
+ connectionId: string | null;
+ apiKeyInfo: unknown;
+ body: unknown;
+ onComplete: ((payload: StreamCompletePayload) => void) | null;
+ onFailure: ((payload: StreamFailurePayload) => void | Promise) | null;
+
+ clientExpectsResponsesStream: boolean;
+ clientExpectsClaudeStream: boolean;
+ shouldEmitDoneTerminator: boolean;
+ expectsOpenAIUsageOnlyChunk: boolean;
+ signatureNamespace: string | null;
+
+ buffer: string;
+ usage: UsageTokenRecord | null;
+ passthroughHasToolCalls: boolean;
+ passthroughToolCalls: Map;
+ passthroughToolCallSeq: number;
+ allowedToolNames: string[];
+ skipPassthroughEvent: boolean;
+ state: TranslateState | null;
+ totalContentLength: number;
+ passthroughAccumulatedContent: string;
+ passthroughAccumulatedReasoning: string;
+ passthroughBufferedTextualToolCallContent: string;
+ passthroughResponsesOutputItems: unknown[];
+ passthroughResponsesPendingFunctionCalls: Map;
+ passthroughResponsesId: string | null;
+ passthroughResponsesCurrentFunctionCallKey: string | null;
+ passthroughResponsesReasoningSummarySeen: Set;
+ streamStartedAt: number;
+ lastToolCallChunkTime: number | null;
+ toolFinishTime: number | null;
+ contentAfterToolSeen: boolean;
+ sessionId: string;
+ pendingToolFinishTime: number | null;
+ doneSent: boolean;
+ pendingPassthroughEventLine: string | null;
+ pendingPassthroughEventEmitted: boolean;
+ lastChunkTime: number;
+ streamTimedOut: boolean;
+
+ decoder: TextDecoder;
+ encoder: TextEncoder;
+ idleTimer: ReturnType | null;
+ claudeEmptyResponseLifecycle: Record;
+ providerPayloadCollector: {
+ push: (v: unknown) => void;
+ build: (...args: unknown[]) => unknown;
+ getEvents: () => unknown[];
+ };
+ clientPayloadCollector: {
+ push: (v: unknown) => void;
+ build: (...args: unknown[]) => unknown;
+ getEvents: () => unknown[];
+ };
+ requestRecord: JsonRecord;
+ requestStreamOptions: JsonRecord;
+};
diff --git a/open-sse/utils/stream/utils.ts b/open-sse/utils/stream/utils.ts
new file mode 100644
index 00000000000..ac58a57aea7
--- /dev/null
+++ b/open-sse/utils/stream/utils.ts
@@ -0,0 +1,54 @@
+import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts";
+import { v4 as uuidv4 } from "uuid";
+
+import { createSSEStream } from "./streamCore.ts";
+import { JsonRecord } from "./types.ts";
+
+/**
+ * Race a response body read against a timeout.
+ * Prevents indefinite hangs when the upstream sends headers but stalls on the body.
+ */
+export function withBodyTimeout(
+ promise: Promise,
+ timeoutMs: number = FETCH_BODY_TIMEOUT_MS
+): Promise {
+ if (timeoutMs <= 0) return promise;
+ let timer: ReturnType;
+ const timeout = new Promise((_, reject) => {
+ timer = setTimeout(() => {
+ const err = new Error(`Response body read timeout after ${timeoutMs}ms`);
+ err.name = "BodyTimeoutError";
+ reject(err);
+ }, timeoutMs);
+ });
+ return Promise.race([promise, timeout]).finally(() => clearTimeout(timer)) as Promise;
+}
+
+export function stringifyIdValue(value: unknown): string | null {
+ return value === null || value === undefined ? null : String(value);
+}
+
+export function asRecord(value: unknown): JsonRecord {
+ return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {};
+}
+
+export const STREAM_SUMMARY_TEXT_LIMIT = 64 * 1024;
+
+export function appendBoundedText(current: string, next: string): string {
+ if (!next) return current;
+ const combined = current + next;
+ if (combined.length <= STREAM_SUMMARY_TEXT_LIMIT) return combined;
+ return combined.slice(-STREAM_SUMMARY_TEXT_LIMIT);
+}
+
+// Note: TextDecoder/TextEncoder are created per-stream inside createSSEStream()
+// to avoid shared state issues with concurrent streams (TextDecoder with {stream:true}
+// maintains internal buffering state between decode() calls).
+
+/**
+ * Stream modes
+ */
+export const STREAM_MODE = {
+ TRANSLATE: "translate", // Full translation between formats
+ PASSTHROUGH: "passthrough", // No translation, normalize output, extract usage
+};
\ No newline at end of file
diff --git a/package-lock.json b/package-lock.json
index e053269520d..8176e182a1e 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "omniroute",
- "version": "3.8.25",
+ "version": "3.8.26",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "omniroute",
- "version": "3.8.25",
+ "version": "3.8.26",
"hasInstallScript": true,
"license": "MIT",
"workspaces": [
@@ -26657,7 +26657,7 @@
},
"open-sse": {
"name": "@omniroute/open-sse",
- "version": "3.8.25"
+ "version": "3.8.26"
}
}
}
diff --git a/package.json b/package.json
index 617e5f72ca1..a5f1cdecca0 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "omniroute",
- "version": "3.8.25",
+ "version": "3.8.26",
"description": "Unified AI router with 160+ providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.",
"type": "module",
"bin": {
diff --git a/scripts/check/check-cognitive-complexity.mjs b/scripts/check/check-cognitive-complexity.mjs
index 90ea5d3e551..663fd8eb441 100644
--- a/scripts/check/check-cognitive-complexity.mjs
+++ b/scripts/check/check-cognitive-complexity.mjs
@@ -31,7 +31,7 @@ const ESLINT_BIN = path.join(ROOT, "node_modules", ".bin", "eslint");
const BASELINE_PATH = path.resolve(
process.argv.includes("--baseline")
? process.argv[process.argv.indexOf("--baseline") + 1]
- : path.join(ROOT, "quality-baseline.json")
+ : path.join(ROOT, "config/quality/quality-baseline.json")
);
const ESLINT_ARGS = [
diff --git a/scripts/check/check-complexity.mjs b/scripts/check/check-complexity.mjs
index bc2f8c42b47..375512bdf84 100644
--- a/scripts/check/check-complexity.mjs
+++ b/scripts/check/check-complexity.mjs
@@ -19,7 +19,7 @@ const ROOT = process.cwd();
const BASELINE_PATH = path.resolve(
process.argv.includes("--baseline")
? process.argv[process.argv.indexOf("--baseline") + 1]
- : path.join(ROOT, "complexity-baseline.json")
+ : path.join(ROOT, "config/quality/complexity-baseline.json")
);
const UPDATE = process.argv.includes("--update");
const CONFIG_PATH = path.join(ROOT, "eslint.complexity.config.mjs");
diff --git a/scripts/check/check-dead-code.mjs b/scripts/check/check-dead-code.mjs
index 3e93890a07f..1c7f5c5e34b 100644
--- a/scripts/check/check-dead-code.mjs
+++ b/scripts/check/check-dead-code.mjs
@@ -28,7 +28,7 @@ const UPDATE = process.argv.includes("--update");
const BASELINE_PATH = path.resolve(
process.argv.includes("--baseline")
? process.argv[process.argv.indexOf("--baseline") + 1]
- : path.join(ROOT, "quality-baseline.json")
+ : path.join(ROOT, "config/quality/quality-baseline.json")
);
/**
diff --git a/scripts/check/check-deps.mjs b/scripts/check/check-deps.mjs
index 1cc88f29c87..cb3eb280964 100644
--- a/scripts/check/check-deps.mjs
+++ b/scripts/check/check-deps.mjs
@@ -29,7 +29,7 @@ import { execFileSync } from "node:child_process";
import { assertNoStale } from "./lib/allowlist.mjs";
const ROOT = process.cwd();
-const ALLOWLIST_PATH = path.join(ROOT, "dependency-allowlist.json");
+const ALLOWLIST_PATH = path.join(ROOT, "config/quality/dependency-allowlist.json");
// Directories to exclude when discovering package.json files.
// Using a set of path segment prefixes (relative to ROOT, forward slashes).
@@ -141,16 +141,12 @@ export function queryNpmRegistry(pkgName, timeoutMs = 8000) {
// Scope packages need URL-encoding for the `npm view` command.
// `npm view` accepts scoped packages natively — no encoding needed.
try {
- const raw = execFileSync(
- "npm",
- ["view", pkgName, "time.created", "--json"],
- {
- encoding: "utf8",
- timeout: timeoutMs,
- // Suppress npm progress/warn output on stderr
- stdio: ["ignore", "pipe", "pipe"],
- }
- );
+ const raw = execFileSync("npm", ["view", pkgName, "time.created", "--json"], {
+ encoding: "utf8",
+ timeout: timeoutMs,
+ // Suppress npm progress/warn output on stderr
+ stdio: ["ignore", "pipe", "pipe"],
+ });
// npm view --json emits a quoted string or null/empty for missing fields
const trimmed = raw.trim();
if (!trimmed) {
@@ -165,7 +161,11 @@ export function queryNpmRegistry(pkgName, timeoutMs = 8000) {
// npm exits with code 1 when the package is NOT found ("E404")
const stderr = err.stderr?.toString() || "";
const stdout = err.stdout?.toString() || "";
- if (stderr.includes("E404") || stdout.includes("E404") || stderr.includes("npm ERR! code E404")) {
+ if (
+ stderr.includes("E404") ||
+ stdout.includes("E404") ||
+ stderr.includes("npm ERR! code E404")
+ ) {
return { exists: false, createdMs: null };
}
// Any other error (ETIMEDOUT, ENOTFOUND, etc.) = network/offline — return null
diff --git a/scripts/check/check-duplication.mjs b/scripts/check/check-duplication.mjs
index 1dc283a59cd..60b5a7bf5ce 100644
--- a/scripts/check/check-duplication.mjs
+++ b/scripts/check/check-duplication.mjs
@@ -15,13 +15,23 @@ const ROOT = process.cwd();
const BASELINE_PATH = path.resolve(
process.argv.includes("--baseline")
? process.argv[process.argv.indexOf("--baseline") + 1]
- : path.join(ROOT, "duplication-baseline.json")
+ : path.join(ROOT, "config/quality/duplication-baseline.json")
);
const UPDATE = process.argv.includes("--update");
const EPS = 0.05; // tolerância de ruído de float (jscpd é determinístico; isto é margem)
// Use local binary (pinned in package.json devDependencies — no registry download at CI time)
const JSCPD_BIN = path.join(ROOT, "node_modules", ".bin", "jscpd");
-const JSCPD_FIXED_ARGS = ["src", "open-sse", "--reporters", "json", "--silent", "--min-tokens", "50", "--ignore", "**/*.test.ts,**/*.test.tsx,**/__tests__/**"];
+const JSCPD_FIXED_ARGS = [
+ "src",
+ "open-sse",
+ "--reporters",
+ "json",
+ "--silent",
+ "--min-tokens",
+ "50",
+ "--ignore",
+ "**/*.test.ts,**/*.test.tsx,**/__tests__/**",
+];
/** Avalia a % atual contra o baseline. */
export function evaluateDuplication(current, baseline, eps = EPS) {
diff --git a/scripts/check/check-fabricated-docs.mjs b/scripts/check/check-fabricated-docs.mjs
index 6bac4aa4d9a..cedc6759780 100644
--- a/scripts/check/check-fabricated-docs.mjs
+++ b/scripts/check/check-fabricated-docs.mjs
@@ -273,8 +273,8 @@ const ENV_VAR_DENYLIST = new Set([
// Gate allowlist constant names (JS identifiers, not env vars) — documented in
// docs/architecture/QUALITY_GATES.md and docs/research/DISCOVERY_TOOL_DESIGN.md
"KNOWN_STALE_DOC_REFS", // export const in check-docs-symbols.mjs
- "KNOWN_MISSING", // export const in check-fetch-targets.mjs
- "KNOWN_RAW_SQL", // export const in check-db-rules.mjs
+ "KNOWN_MISSING", // export const in check-fetch-targets.mjs
+ "KNOWN_RAW_SQL", // export const in check-db-rules.mjs
]);
/** Endpoints that don't follow the standard route.ts pattern. */
@@ -313,6 +313,9 @@ const SKIP_DOC_FILES = new Set([
"docs/reference/PROVIDER_REFERENCE.md", // auto-generated from providers.ts
"docs/reference/openapi.yaml",
"docs/i18n", // translations — separate workflow
+ // Point-in-time documentation audit (v3.8.24): intentionally references drift,
+ // counts, and not-yet-existing files as part of documenting them — not living docs.
+ "docs/ops/DOCUMENTATION_AUDIT_REPORT.md",
]);
// ── File discovery ─────────────────────────────────────────────────────────
diff --git a/scripts/check/check-file-size.mjs b/scripts/check/check-file-size.mjs
index eabc4c68075..82e4193a2a6 100644
--- a/scripts/check/check-file-size.mjs
+++ b/scripts/check/check-file-size.mjs
@@ -15,7 +15,9 @@ function getArg(name, fallback) {
const i = process.argv.indexOf(name);
return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback;
}
-const BASELINE_PATH = path.resolve(getArg("--baseline", path.join(ROOT, "file-size-baseline.json")));
+const BASELINE_PATH = path.resolve(
+ getArg("--baseline", path.join(ROOT, "config/quality/file-size-baseline.json"))
+);
const UPDATE = process.argv.includes("--update");
const SCAN_DIRS = ["src", "open-sse", "electron", "bin"];
// Directories to skip when walking — build artifacts and installed packages.
@@ -30,7 +32,8 @@ export function evaluateFileSizes(currentLocByFile, frozen, cap) {
const improvements = [];
for (const [file, loc] of Object.entries(currentLocByFile)) {
if (file in frozen) {
- if (loc > frozen[file]) violations.push(`${file}: ${loc} > congelado ${frozen[file]} (não pode crescer)`);
+ if (loc > frozen[file])
+ violations.push(`${file}: ${loc} > congelado ${frozen[file]} (não pode crescer)`);
else if (loc < frozen[file]) improvements.push([file, loc]);
} else if (loc > cap) {
violations.push(`${file}: ${loc} > cap ${cap} (arquivo novo acima do limite)`);
@@ -49,7 +52,11 @@ function walk(dir, acc = []) {
const p = path.join(dir, e.name);
if (e.isDirectory()) {
if (!SKIP_DIRS.has(e.name)) walk(p, acc);
- } else if (/\.(ts|tsx)$/.test(e.name) && !/\.test\.tsx?$/.test(e.name) && !/\.d\.ts$/.test(e.name)) {
+ } else if (
+ /\.(ts|tsx)$/.test(e.name) &&
+ !/\.test\.tsx?$/.test(e.name) &&
+ !/\.d\.ts$/.test(e.name)
+ ) {
acc.push(p);
}
}
@@ -59,7 +66,8 @@ function walk(dir, acc = []) {
function collectLoc() {
const out = {};
for (const d of SCAN_DIRS)
- for (const f of walk(path.join(ROOT, d))) out[path.relative(ROOT, f).replace(/\\/g, "/")] = countLines(f);
+ for (const f of walk(path.join(ROOT, d)))
+ out[path.relative(ROOT, f).replace(/\\/g, "/")] = countLines(f);
return out;
}
@@ -76,7 +84,8 @@ function main() {
if (UPDATE && violations.length === 0 && improvements.length) {
for (const [file, loc] of improvements) {
- if (loc <= cap) delete frozen[file]; // caiu para dentro do cap → sai do baseline
+ if (loc <= cap)
+ delete frozen[file]; // caiu para dentro do cap → sai do baseline
else frozen[file] = loc; // continua grande mas encolheu → trava no novo valor
}
baseline.frozen = Object.fromEntries(Object.entries(frozen).sort());
diff --git a/scripts/check/check-licenses.mjs b/scripts/check/check-licenses.mjs
index 3061314db6c..3c265144bcf 100644
--- a/scripts/check/check-licenses.mjs
+++ b/scripts/check/check-licenses.mjs
@@ -22,7 +22,7 @@ import path from "node:path";
import { pathToFileURL } from "node:url";
const ROOT = process.cwd();
-const ALLOWLIST_PATH = path.join(ROOT, ".license-allowlist.json");
+const ALLOWLIST_PATH = path.join(ROOT, "config/quality/.license-allowlist.json");
const CHECKER_BIN = path.join(ROOT, "node_modules", ".bin", "license-checker-rseidelsohn");
const VERBOSE = process.argv.includes("--verbose");
@@ -39,7 +39,9 @@ const PRINT_JSON = process.argv.includes("--json");
*/
export function loadAllowlist() {
if (!fs.existsSync(ALLOWLIST_PATH)) {
- throw new Error(`Allowlist not found: ${ALLOWLIST_PATH}. Create .license-allowlist.json first.`);
+ throw new Error(
+ `Allowlist not found: ${ALLOWLIST_PATH}. Create .license-allowlist.json first.`
+ );
}
const raw = fs.readFileSync(ALLOWLIST_PATH, "utf-8");
const parsed = JSON.parse(raw);
@@ -86,7 +88,10 @@ export function classifyLicense(packageName, license, allowlist) {
}
// 4. Denied
- return { status: "denied", reason: `license '${license}' not in allowlist and no exception registered for '${baseName}'` };
+ return {
+ status: "denied",
+ reason: `license '${license}' not in allowlist and no exception registered for '${baseName}'`,
+ };
}
/**
@@ -195,7 +200,9 @@ function main() {
// Print exceptions (informational)
if (exceptions.length > 0) {
- console.log("\n[check-licenses] Exceções registradas (não bloqueantes, revisar periodicamente):");
+ console.log(
+ "\n[check-licenses] Exceções registradas (não bloqueantes, revisar periodicamente):"
+ );
for (const { pkgKey, license } of exceptions) {
const baseName = stripVersion(pkgKey);
const exc = allowlist.exceptions[baseName];
@@ -217,7 +224,9 @@ function main() {
// Print violations and fail
if (violations.length > 0) {
- console.error("\n[check-licenses] ❌ VIOLAÇÕES DE POLÍTICA — deps de produção com licença não permitida:");
+ console.error(
+ "\n[check-licenses] ❌ VIOLAÇÕES DE POLÍTICA — deps de produção com licença não permitida:"
+ );
for (const { pkgKey, license, reason } of violations) {
console.error(` ✗ ${pkgKey}: ${license}`);
console.error(` → ${reason}`);
@@ -231,11 +240,14 @@ function main() {
return;
}
- console.log("\n[check-licenses] ✅ Todos os pacotes de produção estão em conformidade com a política de licenças.");
+ console.log(
+ "\n[check-licenses] ✅ Todos os pacotes de produção estão em conformidade com a política de licenças."
+ );
}
// Run only when invoked directly (not when imported by tests)
-const isMain = process.argv[1] === pathToFileURL(import.meta.url).pathname ||
+const isMain =
+ process.argv[1] === pathToFileURL(import.meta.url).pathname ||
process.argv[1]?.endsWith("check-licenses.mjs");
if (isMain) {
diff --git a/scripts/check/check-test-discovery.mjs b/scripts/check/check-test-discovery.mjs
index 99948e6a2d0..0d0f19d87b1 100644
--- a/scripts/check/check-test-discovery.mjs
+++ b/scripts/check/check-test-discovery.mjs
@@ -34,7 +34,7 @@ const ROOT = process.cwd();
const BASELINE_PATH = path.resolve(
process.argv.includes("--baseline")
? process.argv[process.argv.indexOf("--baseline") + 1]
- : path.join(ROOT, "test-discovery-baseline.json")
+ : path.join(ROOT, "config/quality/test-discovery-baseline.json")
);
const UPDATE = process.argv.includes("--update");
diff --git a/scripts/check/check-tracked-artifacts.mjs b/scripts/check/check-tracked-artifacts.mjs
index 833ed8dc959..a0324fc168c 100644
--- a/scripts/check/check-tracked-artifacts.mjs
+++ b/scripts/check/check-tracked-artifacts.mjs
@@ -15,7 +15,10 @@ import { execFileSync } from "node:child_process";
import { pathToFileURL } from "node:url";
const FORBIDDEN_PREFIXES = ["node_modules/", ".next/", "coverage/"];
-const FORBIDDEN_EXACT = new Set(["quality-metrics.json"]);
+const FORBIDDEN_EXACT = new Set([
+ "quality-metrics.json", // legacy root location (still forbidden if a stale run writes it)
+ "config/quality/quality-metrics.json", // current generated location (collect-metrics.mjs)
+]);
/**
* Verifica se algum caminho na lista de arquivos rastreados corresponde a um
@@ -80,7 +83,9 @@ function main() {
process.exit(0);
}
- console.error(`[tracked-artifacts] FAIL — ${violations.length} forbidden artifact(s) tracked by git:`);
+ console.error(
+ `[tracked-artifacts] FAIL — ${violations.length} forbidden artifact(s) tracked by git:`
+ );
for (const v of violations) {
console.error(` ✗ ${v}`);
}
diff --git a/scripts/check/check-type-coverage.mjs b/scripts/check/check-type-coverage.mjs
index f6291c77bad..0051063fd35 100644
--- a/scripts/check/check-type-coverage.mjs
+++ b/scripts/check/check-type-coverage.mjs
@@ -33,7 +33,7 @@ const UPDATE = process.argv.includes("--update");
const BASELINE_PATH = path.resolve(
process.argv.includes("--baseline")
? process.argv[process.argv.indexOf("--baseline") + 1]
- : path.join(ROOT, "quality-baseline.json")
+ : path.join(ROOT, "config/quality/quality-baseline.json")
);
// Small epsilon to absorb float noise between runs (type-coverage can vary ~0.01%).
diff --git a/scripts/docs/sync-wiki.mjs b/scripts/docs/sync-wiki.mjs
new file mode 100644
index 00000000000..ff8f6529dd7
--- /dev/null
+++ b/scripts/docs/sync-wiki.mjs
@@ -0,0 +1,290 @@
+#!/usr/bin/env node
+// scripts/docs/sync-wiki.mjs
+// Full GitHub wiki content + cover-count sync.
+//
+// WHY: the wiki has no generator and historically drifts (it sat at "212+ providers /
+// 14 strategies / 37 MCP tools" while code was at 226 / 15 / 87, and new docs like
+// SUPPLY_CHAIN never appeared). This closes the loop: content + counts, automated per
+// release by .github/workflows/wiki-sync.yml.
+//
+// DESIGN — update-in-place, never duplicates:
+// 1. The wiki page names are hand-curated and NOT deterministically reproducible
+// (e.g. "API-Reference" vs "Fly-io-Deployment-Guide"). So we iterate the EXISTING
+// wiki pages and fuzzy-match each to a docs/ source by normalized key
+// (lowercase, strip non-alphanumeric). When a source exists we rewrite that exact
+// page → zero risk of creating a parallel/duplicate page.
+// 2. A curated allowlist (NEW_PAGE_EXCLUDE) keeps internal docs (audit reports, plans,
+// the docs index) off the public wiki; every other unmatched docs page is ADDED
+// with a deterministic acronym-aware name.
+// 3. Hand-curated pages with no docs source (Home, _Sidebar, Header, _Footer,
+// Languages) are left untouched — except the four cover counts on Home.md.
+// 4. EN by default. Localized mirrors (‐Page) are pure update-in-place and
+// only touched with --include-i18n (the i18n source lags and is validated
+// separately).
+//
+// Content transform: strip the YAML frontmatter, prepend the wiki language banner.
+//
+// Usage:
+// node scripts/docs/sync-wiki.mjs --wiki-dir # write
+// node scripts/docs/sync-wiki.mjs --wiki-dir --dry-run # report only
+// node scripts/docs/sync-wiki.mjs --wiki-dir --check # exit 1 on drift
+// node scripts/docs/sync-wiki.mjs --wiki-dir --include-i18n # also localized
+
+import fs from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+const __dirname = path.dirname(fileURLToPath(import.meta.url));
+const ROOT = path.resolve(__dirname, "..", "..");
+
+// U+2010 HYPHEN separates the locale prefix in localized wiki page names.
+const LOCALE_SEP = "‐";
+export const WIKI_BANNER = "> 🌍 [View in other languages](Languages)\n\n\n";
+
+// Docs that must never become public wiki pages (internal reports/plans/index).
+export const NEW_PAGE_EXCLUDE = new Set([
+ "README", // docs index, not a page
+ "DOCUMENTATION_AUDIT_REPORT",
+ "DOCUMENTATION_OVERHAUL_PLAN",
+ "E2E_DASHBOARD_SHAKEDOWN_v3.8.0",
+ "SUBMIT_PR",
+ "fix-opencode-context",
+ "plugins", // docs/dev/plugins.md — internal dev note
+ "SOCKET_DEV_FINDINGS",
+]);
+
+// Acronyms kept upper-case when minting a NEW page name (existing pages keep their
+// curated name via fuzzy match, so this only affects brand-new pages).
+const ACRONYMS = new Set([
+ "api", "mcp", "a2a", "acp", "cli", "sse", "i18n", "pii", "oauth", "vm", "ai",
+ "llm", "sdk", "ide", "ui", "ux", "tls", "mitm", "ws", "cors", "jwt", "db", "vps",
+]);
+
+/** Normalized matching key: lowercase, drop extension + every non-alphanumeric char. */
+export function normKey(s) {
+ return s.toLowerCase().replace(/\.md$/, "").replace(/[^a-z0-9]/g, "");
+}
+
+/** Deterministic wiki page name for a brand-new page (acronym-aware Title-Case-dashed). */
+export function toWikiName(basename) {
+ return basename
+ .replace(/\.md$/, "")
+ .split(/[_\-\s]+/)
+ .filter(Boolean)
+ .map((t) => (ACRONYMS.has(t.toLowerCase()) ? t.toUpperCase() : t[0].toUpperCase() + t.slice(1).toLowerCase()))
+ .join("-");
+}
+
+/** Strip YAML frontmatter and prepend the wiki language banner. Pure; exported for tests. */
+export function toWikiContent(docMarkdown) {
+ const body = docMarkdown.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n/, "").replace(/^\s+/, "");
+ return WIKI_BANNER + body.replace(/\s*$/, "") + "\n";
+}
+
+function read(rel) {
+ const p = path.join(ROOT, rel);
+ return fs.existsSync(p) ? fs.readFileSync(p, "utf8") : "";
+}
+
+// ---- cover-page counts (source of truth) ----
+function providerCount() {
+ const m = read("docs/reference/PROVIDER_REFERENCE.md").match(/Total providers:\s*\*\*(\d+)\*\*/);
+ return m ? Number(m[1]) : null;
+}
+function strategyCount() {
+ const m = read("src/shared/constants/routingStrategies.ts").match(
+ /ROUTING_STRATEGY_VALUES\s*=\s*\[([^\]]*)\]/
+ );
+ return m ? (m[1].match(/"[^"]+"/g) || []).length : null;
+}
+function localeCount() {
+ try {
+ const c = JSON.parse(read("config/i18n.json"));
+ return Array.isArray(c.locales) ? c.locales.length : null;
+ } catch {
+ return null;
+ }
+}
+function mcpToolCount() {
+ // Prefer a literal; the constant is computed at runtime so best-effort only.
+ const m = read("open-sse/mcp-server/server.ts").match(/TOTAL_MCP_TOOL_COUNT\s*=\s*(\d+)\b/);
+ return m ? Number(m[1]) : null;
+}
+export function readCounts() {
+ return {
+ providers: providerCount(),
+ strategies: strategyCount(),
+ mcpTools: mcpToolCount(),
+ locales: localeCount(),
+ };
+}
+
+/** Apply cover-page count substitutions to Home.md text. Pure; exported for tests. */
+export function syncHomeCounts(home, counts) {
+ let out = home;
+ if (counts.providers) {
+ out = out
+ .replace(/Connect every AI tool to \d+ providers/g, `Connect every AI tool to ${counts.providers} providers`)
+ .replace(/\*\*\d+ AI Providers\*\*/g, `**${counts.providers} AI Providers**`)
+ .replace(/All \d+ supported providers/g, `All ${counts.providers} supported providers`)
+ .replace(/\b\d+ providers\b/g, `${counts.providers} providers`);
+ }
+ if (counts.strategies) {
+ out = out.replace(/\*\*\d+ Routing Strategies\*\*/g, `**${counts.strategies} Routing Strategies**`);
+ }
+ if (counts.mcpTools) {
+ out = out.replace(/(\|\s*\*\*MCP Server\*\*\s*\|\s*)\d+( tools)/g, `$1${counts.mcpTools}$2`);
+ }
+ return out;
+}
+
+// ---- docs discovery ----
+function walkMarkdown(dir, acc = []) {
+ if (!fs.existsSync(dir)) return acc;
+ for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
+ const p = path.join(dir, e.name);
+ if (e.isDirectory()) walkMarkdown(p, acc);
+ else if (e.name.endsWith(".md")) acc.push(p);
+ }
+ return acc;
+}
+
+/** Build normKey → docs absolute path for English docs (docs/ minus docs/i18n). */
+function indexEnglishDocs() {
+ const docsRoot = path.join(ROOT, "docs");
+ const files = walkMarkdown(docsRoot).filter((f) => !f.includes(`${path.sep}i18n${path.sep}`));
+ const byKey = new Map();
+ for (const f of files) {
+ const base = path.basename(f, ".md");
+ const k = normKey(base);
+ // First-writer-wins keeps a deterministic pick for basename collisions.
+ if (!byKey.has(k)) byKey.set(k, { file: f, base });
+ }
+ return byKey;
+}
+
+/** Build normKey → docs path for a given locale's i18n tree. */
+function indexLocaleDocs(locale) {
+ const root = path.join(ROOT, "docs", "i18n", locale);
+ const byKey = new Map();
+ for (const f of walkMarkdown(root)) {
+ const k = normKey(path.basename(f, ".md"));
+ if (!byKey.has(k)) byKey.set(k, f);
+ }
+ return byKey;
+}
+
+function listWikiPages(wikiDir) {
+ return fs
+ .readdirSync(wikiDir)
+ .filter((n) => n.endsWith(".md"))
+ .map((n) => n.slice(0, -3));
+}
+
+/** Split a wiki page name into { locale, name }. EN pages have locale = null. */
+export function parseWikiPage(page) {
+ const idx = page.indexOf(LOCALE_SEP);
+ if (idx === -1) return { locale: null, name: page };
+ return { locale: page.slice(0, idx), name: page.slice(idx + 1) };
+}
+
+function main() {
+ const args = process.argv.slice(2);
+ const wikiDir = args.includes("--wiki-dir") ? args[args.indexOf("--wiki-dir") + 1] : null;
+ const dryRun = args.includes("--dry-run");
+ const check = args.includes("--check");
+ const includeI18n = args.includes("--include-i18n");
+ const updateExisting = args.includes("--update-existing");
+ if (!wikiDir || !fs.existsSync(wikiDir)) {
+ console.error("usage: sync-wiki.mjs --wiki-dir [--dry-run|--check] [--include-i18n]");
+ process.exit(2);
+ }
+
+ const enDocs = indexEnglishDocs();
+ const wikiPages = listWikiPages(wikiDir);
+ const enWikiKeys = new Set();
+ const localeIndexes = new Map();
+
+ const plan = { update: [], add: [], untouched: [], countsChanged: false };
+
+ // 1. Update existing wiki pages from their docs source.
+ for (const page of wikiPages) {
+ if (page === "Home") continue; // handled by counts below
+ const { locale, name } = parseWikiPage(page);
+ const key = normKey(name);
+ let srcFile = null;
+ if (!locale) {
+ enWikiKeys.add(key);
+ srcFile = enDocs.get(key)?.file ?? null;
+ } else if (includeI18n) {
+ if (!localeIndexes.has(locale)) localeIndexes.set(locale, indexLocaleDocs(locale));
+ srcFile = localeIndexes.get(locale).get(key) ?? null;
+ }
+ if (!srcFile) {
+ plan.untouched.push(page);
+ continue;
+ }
+ const next = toWikiContent(fs.readFileSync(srcFile, "utf8"));
+ const cur = fs.readFileSync(path.join(wikiDir, `${page}.md`), "utf8");
+ if (next !== cur) plan.update.push({ page, srcFile });
+ }
+
+ // 2. Add curated new English pages (unmatched docs, minus the exclude list).
+ for (const [key, { file, base }] of enDocs) {
+ if (enWikiKeys.has(key)) continue;
+ if (NEW_PAGE_EXCLUDE.has(base)) continue;
+ plan.add.push({ page: toWikiName(base), srcFile: file, base });
+ }
+
+ // 3. Home cover counts.
+ const counts = readCounts();
+ const homePath = path.join(wikiDir, "Home.md");
+ let homeAfter = null;
+ if (fs.existsSync(homePath)) {
+ const before = fs.readFileSync(homePath, "utf8");
+ homeAfter = syncHomeCounts(before, counts);
+ plan.countsChanged = homeAfter !== before;
+ }
+
+ // ---- report ----
+ // Updating existing pages is opt-in: several docs SOURCES carry stale counts (e.g.
+ // ARCHITECTURE.md still says "177 providers / 37 MCP tools" while the wiki cover was
+ // hand-patched to 226/87). Overwriting from a staler source would REGRESS the wiki, so
+ // by default we only ADD missing pages and sync Home counts. Pass --update-existing
+ // once the docs sources are regenerated (see docs/ops/DOCUMENTATION_AUDIT_REPORT.md).
+ const updates = updateExisting ? plan.update : [];
+ const total = updates.length + plan.add.length + (plan.countsChanged ? 1 : 0);
+ console.log(`[wiki-sync] counts: ${JSON.stringify(counts)}`);
+ console.log(
+ `[wiki-sync] add: ${plan.add.length} | Home counts: ${plan.countsChanged ? "drift" : "in-sync"} | ` +
+ `existing-page updates: ${plan.update.length} (${updateExisting ? "ENABLED" : "skipped — needs --update-existing"}) | untouched: ${plan.untouched.length}`
+ );
+ if (dryRun || check) {
+ if (plan.add.length) console.log(` add → ${plan.add.map((a) => a.page).join(", ")}`);
+ if (plan.update.length)
+ console.log(
+ ` ${updateExisting ? "update" : "would-update (skipped)"} → ${plan.update.map((u) => u.page).slice(0, 60).join(", ")}${plan.update.length > 60 ? " …" : ""}`
+ );
+ if (check) {
+ if (total > 0) {
+ console.error(`✗ wiki out of sync (${total} change(s) pending)`);
+ process.exit(1);
+ }
+ console.log("✓ wiki in sync");
+ }
+ return;
+ }
+
+ // ---- write ----
+ for (const { page, srcFile } of [...updates, ...plan.add]) {
+ fs.writeFileSync(path.join(wikiDir, `${page}.md`), toWikiContent(fs.readFileSync(srcFile, "utf8")));
+ }
+ if (plan.countsChanged && homeAfter != null) fs.writeFileSync(homePath, homeAfter);
+ console.log(
+ `[wiki-sync] wrote ${total} page(s) (add: ${plan.add.length}, updates: ${updates.length}, counts: ${plan.countsChanged ? 1 : 0}).`
+ );
+}
+
+const invokedDirectly =
+ process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
+if (invokedDirectly) main();
diff --git a/scripts/quality/check-quality-ratchet.mjs b/scripts/quality/check-quality-ratchet.mjs
index 2e885199078..9948c5bb5a9 100644
--- a/scripts/quality/check-quality-ratchet.mjs
+++ b/scripts/quality/check-quality-ratchet.mjs
@@ -12,8 +12,12 @@ function getArg(name, fallback) {
const i = process.argv.indexOf(name);
return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback;
}
-const BASELINE = path.resolve(getArg("--baseline", path.join(cwd, "quality-baseline.json")));
-const METRICS = path.resolve(getArg("--metrics", path.join(cwd, "quality-metrics.json")));
+const BASELINE = path.resolve(
+ getArg("--baseline", path.join(cwd, "config/quality/quality-baseline.json"))
+);
+const METRICS = path.resolve(
+ getArg("--metrics", path.join(cwd, "config/quality/quality-metrics.json"))
+);
const SUMMARY = getArg("--summary", null);
const UPDATE = process.argv.includes("--update");
// --allow-missing: pula métricas do baseline ausentes do metrics (em vez de falhar).
@@ -73,7 +77,7 @@ for (const [key, spec] of Object.entries(baseline.metrics)) {
status = "↑ melhorou";
if (REQUIRE_TIGHTEN && base - current > tightenSlack) {
tightenFailures.push(
- `${key}: melhorou de ${base} para ${current} (delta ${(base - current).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR`,
+ `${key}: melhorou de ${base} para ${current} (delta ${(base - current).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR`
);
}
}
@@ -86,7 +90,7 @@ for (const [key, spec] of Object.entries(baseline.metrics)) {
status = "↑ melhorou";
if (REQUIRE_TIGHTEN && current - base > tightenSlack) {
tightenFailures.push(
- `${key}: melhorou de ${base} para ${current} (delta ${(current - base).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR`,
+ `${key}: melhorou de ${base} para ${current} (delta ${(current - base).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR`
);
}
}
@@ -99,10 +103,10 @@ const baselineKeys = new Set(Object.keys(baseline.metrics));
const orphans = Object.keys(metrics).filter((k) => !baselineKeys.has(k));
if (orphans.length > 0) {
console.warn(
- `[quality-ratchet] WARN: ${orphans.length} métrica(s) órfã(s) — presente(s) em ${path.basename(METRICS)} mas sem entrada no baseline: ${orphans.join(", ")}`,
+ `[quality-ratchet] WARN: ${orphans.length} métrica(s) órfã(s) — presente(s) em ${path.basename(METRICS)} mas sem entrada no baseline: ${orphans.join(", ")}`
);
console.warn(
- `[quality-ratchet] WARN: adicione ${orphans.length === 1 ? "essa métrica" : "essas métricas"} ao baseline (com value/direction) para que sejam catraceadas.`,
+ `[quality-ratchet] WARN: adicione ${orphans.length === 1 ? "essa métrica" : "essas métricas"} ao baseline (com value/direction) para que sejam catraceadas.`
);
}
@@ -140,7 +144,7 @@ if (failures.length) {
if (REQUIRE_TIGHTEN && !UPDATE && tightenFailures.length > 0) {
console.error(
"[quality-ratchet] FALHOU (--require-tighten): métrica(s) melhoraram mas o baseline não foi apertado:\n" +
- tightenFailures.map((f) => " ✗ " + f).join("\n"),
+ tightenFailures.map((f) => " ✗ " + f).join("\n")
);
process.exit(1);
}
diff --git a/scripts/quality/collect-metrics.mjs b/scripts/quality/collect-metrics.mjs
index bf24f96b29b..2807409831e 100644
--- a/scripts/quality/collect-metrics.mjs
+++ b/scripts/quality/collect-metrics.mjs
@@ -250,6 +250,9 @@ if (import.meta.url === pathToFileURL(process.argv[1] || "").href) {
coverageByModule();
openapiCoverage();
await i18nUiCoverage();
- fs.writeFileSync(path.join(cwd, "quality-metrics.json"), JSON.stringify(out, null, 2) + "\n");
+ fs.writeFileSync(
+ path.join(cwd, "config/quality/quality-metrics.json"),
+ JSON.stringify(out, null, 2) + "\n"
+ );
console.log("[collect-metrics]", JSON.stringify(out));
}
diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/OpenRouterPresetInput.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/OpenRouterPresetInput.tsx
new file mode 100644
index 00000000000..280f420f690
--- /dev/null
+++ b/src/app/(dashboard)/dashboard/providers/[id]/components/OpenRouterPresetInput.tsx
@@ -0,0 +1,54 @@
+"use client";
+
+import { useCallback, useMemo, useState } from "react";
+import { Input } from "@/shared/components";
+import { OPENROUTER_PRESET_MAX_LENGTH } from "@/shared/constants/openRouterPreset";
+import { providerText, type ProviderMessageTranslator } from "../providerPageHelpers";
+
+interface OpenRouterPresetInputProps {
+ value: string;
+ onChange: (value: string) => void;
+ t: ProviderMessageTranslator;
+}
+
+export default function OpenRouterPresetInput({ value, onChange, t }: OpenRouterPresetInputProps) {
+ return (
+ onChange(e.target.value)}
+ placeholder="email-copywriter"
+ maxLength={OPENROUTER_PRESET_MAX_LENGTH}
+ hint={providerText(
+ t,
+ "openRouterPresetHint",
+ "Sends this connection's preset as the OpenRouter top-level preset field."
+ )}
+ />
+ );
+}
+
+export function useOpenRouterPresetControl(
+ provider: string | null | undefined,
+ t: ProviderMessageTranslator
+) {
+ const [value, setValue] = useState("");
+ const isOpenRouter = provider === "openrouter";
+ const applyTo = useCallback(
+ (data: Record) => {
+ const preset = value.trim();
+ if (isOpenRouter && preset) data.preset = preset;
+ },
+ [isOpenRouter, value]
+ );
+ const getPatch = useCallback(() => {
+ if (!isOpenRouter) return {};
+ const preset = value.trim();
+ return { preset: preset || null };
+ }, [isOpenRouter, value]);
+ const input = useMemo(
+ () => (isOpenRouter ? : null),
+ [isOpenRouter, t, value]
+ );
+ return { applyTo, getPatch, input, isOpenRouter, setValue };
+}
diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx
index df432143dcd..59d94f0131f 100644
--- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx
+++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx
@@ -1,15 +1,9 @@
"use client";
-// Issue #3501 Phase 1c — extracted from the god-component.
-// ~787-LOC modal for adding a new API key / credential to a provider.
-
import { useState, useEffect, useRef } from "react";
import { useTranslations } from "next-intl";
import { Button, Badge, Input, Modal, Toggle } from "@/shared/components";
-import {
- providerAllowsOptionalApiKey,
- supportsBulkApiKey,
-} from "@/shared/constants/providers";
+import { providerAllowsOptionalApiKey, supportsBulkApiKey } from "@/shared/constants/providers";
import { parseBulkApiKeys } from "@/shared/utils/bulkApiKeyParser";
import {
isBaseUrlConfigurableProvider,
@@ -30,6 +24,7 @@ import {
type CommandCodeAuthFlowState,
} from "../../providerPageHelpers";
import { getWebSessionCredentialRequirement } from "../../webSessionCredentials";
+import { useOpenRouterPresetControl } from "../OpenRouterPresetInput";
import WebSessionCredentialGuide from "../WebSessionCredentialGuide";
export interface AddApiKeyModalProps {
@@ -76,6 +71,7 @@ export default function AddApiKeyModal({
const defaultRegion = isBedrock ? "eu-west-2" : "us-central1";
const isGlm = isGlmProvider(provider);
const isQoder = provider === "qoder";
+ const openRouterPreset = useOpenRouterPresetControl(provider, t);
const isCloudflare = provider === "cloudflare-ai";
const localProviderMetadata = getLocalProviderMetadata(provider);
const isLocalSelfHostedProvider = !!localProviderMetadata;
@@ -133,7 +129,6 @@ export default function AddApiKeyModal({
baseUrl: initialBaseUrl || defaultBaseUrl,
}));
}, [defaultBaseUrl, initialBaseUrl, isOpen]);
-
const bulkSupported = supportsBulkApiKey(provider);
const [mode, setMode] = useState<"single" | "bulk">("single");
const [bulkText, setBulkText] = useState("");
@@ -278,7 +273,6 @@ export default function AddApiKeyModal({
if (!isValid) {
if (apiKeyOptional && !credentialInput) {
- // Bypass validation block for local/optional providers when no key is provided
console.debug("Validation failed but apiKey is optional; proceeding to save.");
} else {
setSaveError(validationError || credentialValidationFailedMessage);
@@ -290,6 +284,7 @@ export default function AddApiKeyModal({
if (formData.customUserAgent.trim()) {
providerSpecificData.customUserAgent = formData.customUserAgent.trim();
}
+ openRouterPreset.applyTo(providerSpecificData);
if (formData.routingTags.trim()) {
providerSpecificData.tags = parseRoutingTagsInput(formData.routingTags);
}
@@ -347,15 +342,18 @@ export default function AddApiKeyModal({
setSaveError(null);
try {
- let providerSpecificData: Record | undefined;
+ const bulkProviderSpecificData: Record = {};
if (usesBaseUrl) {
const checked = normalizeAndValidateHttpBaseUrl(formData.baseUrl, defaultBaseUrl);
if (checked.error) {
setSaveError(checked.error);
return;
}
- providerSpecificData = { baseUrl: checked.value };
+ bulkProviderSpecificData.baseUrl = checked.value;
}
+ openRouterPreset.applyTo(bulkProviderSpecificData);
+ const providerSpecificData =
+ Object.keys(bulkProviderSpecificData).length > 0 ? bulkProviderSpecificData : undefined;
const res = await fetch("/api/providers/bulk", {
method: "POST",
@@ -432,6 +430,7 @@ export default function AddApiKeyModal({
{bulkSupported && mode === "bulk" && (
{t("bulkAddFormatHint")}
+ {openRouterPreset.input}
+ {openRouterPreset.input}
{
if (!connection?.provider) return;
@@ -392,7 +396,6 @@ export default function EditConnectionModal({
healthCheckInterval: formData.healthCheckInterval,
};
- // Build rateLimitOverrides from non-empty fields
const overrides: Record
= {};
if (formData.rpm.trim()) overrides.rpm = Number(formData.rpm);
if (formData.tpm.trim()) overrides.tpm = Number(formData.tpm);
@@ -460,7 +463,6 @@ export default function EditConnectionModal({
updates.rateLimitedUntil = null;
}
}
- // Persist extra API keys and baseUrl in providerSpecificData
if (!isOAuth) {
updates.providerSpecificData = {
...(connection.providerSpecificData || {}),
@@ -469,7 +471,7 @@ export default function EditConnectionModal({
tags: parseRoutingTagsInput(formData.routingTags),
excludedModels: parseExcludedModelsInput(formData.excludedModels),
customUserAgent: formData.customUserAgent.trim(),
- // Only write when explicitly enabled; omit to let registry default take effect
+ ...openRouterPreset.getPatch(),
...(formData.passthroughModels ? { passthroughModels: true } : {}),
};
if (connection.provider === "bailian-coding-plan") {
@@ -513,7 +515,6 @@ export default function EditConnectionModal({
Object.keys(currentRequestDefaults).length > 0 ? currentRequestDefaults : undefined;
}
} else {
- // Also persist tag for OAuth accounts
updates.providerSpecificData = {
...(connection.providerSpecificData || {}),
tag: formData.tag.trim() || undefined,
@@ -546,8 +547,6 @@ export default function EditConnectionModal({
),
};
}
- // #2997: persist the transient-cooldown opt-out; write only when enabled,
- // clear it otherwise so a disabled toggle does not linger as `false`.
if (updates.providerSpecificData) {
updates.providerSpecificData.disableCooling = formData.disableCooling ? true : undefined;
}
@@ -831,6 +830,7 @@ export default function EditConnectionModal({
placeholder="my-app/1.0"
hint={t("customUserAgentHint")}
/>
+ {openRouterPreset.input}
= {
supportsTools: true,
},
+ // ── Z.AI GLM-5.2 (1M context, 128K max output, effort tiers) ────
+ "glm-5.2": {
+ maxOutputTokens: 131072,
+ contextWindow: 1000000,
+ supportsThinking: true,
+ supportsTools: true,
+ },
+ "glm-5.2-high": {
+ maxOutputTokens: 131072,
+ contextWindow: 1000000,
+ supportsThinking: true,
+ supportsTools: true,
+ },
+ "glm-5.2-max": {
+ maxOutputTokens: 131072,
+ contextWindow: 1000000,
+ supportsThinking: true,
+ supportsTools: true,
+ },
+
// ── Z.AI GLM-5.x (200K context, 128K max output) ─────────────────
"glm-5.1": {
maxOutputTokens: 128000,
diff --git a/src/shared/constants/openRouterPreset.ts b/src/shared/constants/openRouterPreset.ts
new file mode 100644
index 00000000000..745ddef85b4
--- /dev/null
+++ b/src/shared/constants/openRouterPreset.ts
@@ -0,0 +1,11 @@
+export const OPENROUTER_PRESET_MAX_LENGTH = 200;
+
+export function isOpenRouterPresetValue(value: unknown): value is string {
+ return typeof value === "string" && value.trim().length <= OPENROUTER_PRESET_MAX_LENGTH;
+}
+
+export function normalizeOpenRouterPreset(value: unknown): string | undefined {
+ if (!isOpenRouterPresetValue(value)) return undefined;
+ const normalized = value.trim();
+ return normalized || undefined;
+}
diff --git a/src/shared/constants/pricing.ts b/src/shared/constants/pricing.ts
index d245c9aab97..392afafd50c 100644
--- a/src/shared/constants/pricing.ts
+++ b/src/shared/constants/pricing.ts
@@ -60,6 +60,27 @@ const CLAUDE_SONNET_46_PRICING = {
};
const GLM_PRICING = {
+ "glm-5.2": {
+ input: 1.2,
+ output: 5,
+ cached: 0.3,
+ reasoning: 5,
+ cache_creation: 1.2,
+ },
+ "glm-5.2-high": {
+ input: 1.2,
+ output: 5,
+ cached: 0.3,
+ reasoning: 5,
+ cache_creation: 1.2,
+ },
+ "glm-5.2-max": {
+ input: 1.2,
+ output: 5,
+ cached: 0.3,
+ reasoning: 5,
+ cache_creation: 1.2,
+ },
"glm-5.1": {
input: 1.2,
output: 5,
diff --git a/src/shared/validation/providerSpecificData.ts b/src/shared/validation/providerSpecificData.ts
new file mode 100644
index 00000000000..138119b8e7e
--- /dev/null
+++ b/src/shared/validation/providerSpecificData.ts
@@ -0,0 +1,265 @@
+import { z } from "zod";
+import {
+ OPENROUTER_PRESET_MAX_LENGTH,
+ isOpenRouterPresetValue,
+} from "@/shared/constants/openRouterPreset";
+
+function isHttpUrl(value: string): boolean {
+ try {
+ const parsed = new URL(value);
+ return parsed.protocol === "http:" || parsed.protocol === "https:";
+ } catch {
+ return false;
+ }
+}
+
+const CODEX_REASONING_EFFORT_VALUES = new Set(["none", "low", "medium", "high", "xhigh"]);
+const REQUEST_DEFAULT_SERVICE_TIER_VALUES = new Set(["default", "priority", "fast", "flex"]);
+
+export function validateProviderSpecificData(
+ data: Record | undefined,
+ ctx: z.RefinementCtx
+): void {
+ if (!data) return;
+
+ const baseUrl = data.baseUrl;
+ if (baseUrl !== undefined && (typeof baseUrl !== "string" || !isHttpUrl(baseUrl))) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.baseUrl must be a valid http(s) URL",
+ path: ["baseUrl"],
+ });
+ }
+
+ const customUserAgent = data.customUserAgent;
+ if (
+ customUserAgent !== undefined &&
+ customUserAgent !== null &&
+ (typeof customUserAgent !== "string" || customUserAgent.length > 500)
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.customUserAgent must be a string up to 500 chars",
+ path: ["customUserAgent"],
+ });
+ }
+
+ const cx = data.cx;
+ if (cx !== undefined && cx !== null && (typeof cx !== "string" || cx.length > 500)) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.cx must be a string up to 500 chars",
+ path: ["cx"],
+ });
+ }
+
+ const region = data.region;
+ if (
+ region !== undefined &&
+ region !== null &&
+ (typeof region !== "string" || region.length > 64)
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.region must be a string up to 64 chars",
+ path: ["region"],
+ });
+ }
+
+ const openaiStoreEnabled = data.openaiStoreEnabled;
+ if (openaiStoreEnabled !== undefined && typeof openaiStoreEnabled !== "boolean") {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.openaiStoreEnabled must be a boolean",
+ path: ["openaiStoreEnabled"],
+ });
+ }
+
+ const blockExtraUsage = data.blockExtraUsage;
+ if (blockExtraUsage !== undefined && typeof blockExtraUsage !== "boolean") {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.blockExtraUsage must be a boolean",
+ path: ["blockExtraUsage"],
+ });
+ }
+
+ const autoFetchModels = data.autoFetchModels;
+ if (autoFetchModels !== undefined && typeof autoFetchModels !== "boolean") {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.autoFetchModels must be a boolean",
+ path: ["autoFetchModels"],
+ });
+ }
+
+ const disableStreamOptions = data.disableStreamOptions;
+ if (disableStreamOptions !== undefined && typeof disableStreamOptions !== "boolean") {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.disableStreamOptions must be a boolean",
+ path: ["disableStreamOptions"],
+ });
+ }
+
+ const preset = data.preset;
+ if (preset !== undefined && preset !== null && !isOpenRouterPresetValue(preset)) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: `providerSpecificData.preset must be a string up to ${OPENROUTER_PRESET_MAX_LENGTH} chars`,
+ path: ["preset"],
+ });
+ }
+
+ const requestDefaults = data.requestDefaults;
+ if (requestDefaults !== undefined) {
+ if (!requestDefaults || typeof requestDefaults !== "object" || Array.isArray(requestDefaults)) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.requestDefaults must be an object",
+ path: ["requestDefaults"],
+ });
+ } else {
+ const requestDefaultsRecord = requestDefaults as Record;
+ const reasoningEffort = requestDefaultsRecord.reasoningEffort;
+ if (
+ reasoningEffort !== undefined &&
+ reasoningEffort !== null &&
+ (typeof reasoningEffort !== "string" ||
+ !CODEX_REASONING_EFFORT_VALUES.has(reasoningEffort.trim().toLowerCase()))
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message:
+ "providerSpecificData.requestDefaults.reasoningEffort must be one of none, low, medium, high, xhigh",
+ path: ["requestDefaults", "reasoningEffort"],
+ });
+ }
+
+ const serviceTier = requestDefaultsRecord.serviceTier;
+ if (
+ serviceTier !== undefined &&
+ serviceTier !== null &&
+ (typeof serviceTier !== "string" ||
+ !REQUEST_DEFAULT_SERVICE_TIER_VALUES.has(serviceTier.trim().toLowerCase()))
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message:
+ "providerSpecificData.requestDefaults.serviceTier must be one of default, priority, fast, flex when provided",
+ path: ["requestDefaults", "serviceTier"],
+ });
+ }
+
+ const context1m = requestDefaultsRecord.context1m;
+ if (context1m !== undefined && context1m !== null && typeof context1m !== "boolean") {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.requestDefaults.context1m must be a boolean",
+ path: ["requestDefaults", "context1m"],
+ });
+ }
+ }
+ }
+
+ const consoleApiKey = data.consoleApiKey;
+ if (consoleApiKey !== undefined && consoleApiKey !== null && typeof consoleApiKey !== "string") {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.consoleApiKey must be a string",
+ path: ["consoleApiKey"],
+ });
+ }
+ if (typeof consoleApiKey === "string" && consoleApiKey.length > 10000) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.consoleApiKey must be at most 10000 characters",
+ path: ["consoleApiKey"],
+ });
+ }
+
+ const groupTag = data.tag;
+ if (
+ groupTag !== undefined &&
+ groupTag !== null &&
+ (typeof groupTag !== "string" || groupTag.length > 100)
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.tag must be a string up to 100 chars",
+ path: ["tag"],
+ });
+ }
+
+ const routingTags = data.tags;
+ if (routingTags !== undefined && routingTags !== null) {
+ if (!Array.isArray(routingTags) || routingTags.length > 50) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.tags must be an array with at most 50 items",
+ path: ["tags"],
+ });
+ } else if (
+ routingTags.some(
+ (tag) => typeof tag !== "string" || tag.trim().length === 0 || tag.trim().length > 64
+ )
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message:
+ "providerSpecificData.tags must contain non-empty strings up to 64 characters each",
+ path: ["tags"],
+ });
+ }
+ }
+
+ const excludedModels = data.excludedModels ?? data.excluded_models;
+ if (excludedModels !== undefined && excludedModels !== null) {
+ if (typeof excludedModels === "string") {
+ if (excludedModels.length > 5000) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.excludedModels string must be up to 5000 chars",
+ path: ["excludedModels"],
+ });
+ }
+ } else if (!Array.isArray(excludedModels) || excludedModels.length > 100) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message: "providerSpecificData.excludedModels must be an array with at most 100 items",
+ path: ["excludedModels"],
+ });
+ } else if (
+ excludedModels.some(
+ (pattern) =>
+ typeof pattern !== "string" ||
+ pattern.trim().length === 0 ||
+ pattern.trim().length > 200 ||
+ pattern.trim() === "**"
+ )
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message:
+ "providerSpecificData.excludedModels must contain non-empty patterns up to 200 characters",
+ path: ["excludedModels"],
+ });
+ }
+ }
+
+ const clientProfile = data.clientProfile;
+ if (clientProfile !== undefined && clientProfile !== null) {
+ const normalized = typeof clientProfile === "string" ? clientProfile.trim().toLowerCase() : "";
+ if (
+ typeof clientProfile !== "string" ||
+ !["ide", "harness", "cli", "sdk"].includes(normalized)
+ ) {
+ ctx.addIssue({
+ code: z.ZodIssueCode.custom,
+ message:
+ "providerSpecificData.clientProfile must be ide, harness, cli, or sdk (cli/sdk map to harness)",
+ path: ["clientProfile"],
+ });
+ }
+ }
+}
diff --git a/src/shared/validation/schemas.ts b/src/shared/validation/schemas.ts
index dc2084a6721..480a01e9c6c 100644
--- a/src/shared/validation/schemas.ts
+++ b/src/shared/validation/schemas.ts
@@ -13,259 +13,7 @@ import {
isForbiddenCustomHeaderName,
} from "@/shared/constants/upstreamHeaders";
import { MAX_TIMER_TIMEOUT_MS } from "@/shared/utils/runtimeTimeouts";
-
-function isHttpUrl(value: string): boolean {
- try {
- const parsed = new URL(value);
- return parsed.protocol === "http:" || parsed.protocol === "https:";
- } catch {
- return false;
- }
-}
-
-const CODEX_REASONING_EFFORT_VALUES = new Set(["none", "low", "medium", "high", "xhigh"]);
-const REQUEST_DEFAULT_SERVICE_TIER_VALUES = new Set(["default", "priority", "fast", "flex"]);
-
-function validateProviderSpecificData(
- data: Record | undefined,
- ctx: z.RefinementCtx
-): void {
- if (!data) return;
-
- const baseUrl = data.baseUrl;
- if (baseUrl !== undefined && (typeof baseUrl !== "string" || !isHttpUrl(baseUrl))) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.baseUrl must be a valid http(s) URL",
- path: ["baseUrl"],
- });
- }
-
- const customUserAgent = data.customUserAgent;
- if (
- customUserAgent !== undefined &&
- customUserAgent !== null &&
- (typeof customUserAgent !== "string" || customUserAgent.length > 500)
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.customUserAgent must be a string up to 500 chars",
- path: ["customUserAgent"],
- });
- }
-
- const cx = data.cx;
- if (cx !== undefined && cx !== null && (typeof cx !== "string" || cx.length > 500)) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.cx must be a string up to 500 chars",
- path: ["cx"],
- });
- }
-
- const region = data.region;
- if (
- region !== undefined &&
- region !== null &&
- (typeof region !== "string" || region.length > 64)
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.region must be a string up to 64 chars",
- path: ["region"],
- });
- }
-
- const openaiStoreEnabled = data.openaiStoreEnabled;
- if (openaiStoreEnabled !== undefined && typeof openaiStoreEnabled !== "boolean") {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.openaiStoreEnabled must be a boolean",
- path: ["openaiStoreEnabled"],
- });
- }
-
- const blockExtraUsage = data.blockExtraUsage;
- if (blockExtraUsage !== undefined && typeof blockExtraUsage !== "boolean") {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.blockExtraUsage must be a boolean",
- path: ["blockExtraUsage"],
- });
- }
-
- const autoFetchModels = data.autoFetchModels;
- if (autoFetchModels !== undefined && typeof autoFetchModels !== "boolean") {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.autoFetchModels must be a boolean",
- path: ["autoFetchModels"],
- });
- }
-
- const disableStreamOptions = data.disableStreamOptions;
- if (disableStreamOptions !== undefined && typeof disableStreamOptions !== "boolean") {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.disableStreamOptions must be a boolean",
- path: ["disableStreamOptions"],
- });
- }
-
- const requestDefaults = data.requestDefaults;
- if (requestDefaults !== undefined) {
- if (!requestDefaults || typeof requestDefaults !== "object" || Array.isArray(requestDefaults)) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.requestDefaults must be an object",
- path: ["requestDefaults"],
- });
- } else {
- const requestDefaultsRecord = requestDefaults as Record;
- const reasoningEffort = requestDefaultsRecord.reasoningEffort;
- if (
- reasoningEffort !== undefined &&
- reasoningEffort !== null &&
- (typeof reasoningEffort !== "string" ||
- !CODEX_REASONING_EFFORT_VALUES.has(reasoningEffort.trim().toLowerCase()))
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message:
- "providerSpecificData.requestDefaults.reasoningEffort must be one of none, low, medium, high, xhigh",
- path: ["requestDefaults", "reasoningEffort"],
- });
- }
-
- const serviceTier = requestDefaultsRecord.serviceTier;
- if (
- serviceTier !== undefined &&
- serviceTier !== null &&
- (typeof serviceTier !== "string" ||
- !REQUEST_DEFAULT_SERVICE_TIER_VALUES.has(serviceTier.trim().toLowerCase()))
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message:
- "providerSpecificData.requestDefaults.serviceTier must be one of default, priority, fast, flex when provided",
- path: ["requestDefaults", "serviceTier"],
- });
- }
-
- const context1m = requestDefaultsRecord.context1m;
- if (context1m !== undefined && context1m !== null && typeof context1m !== "boolean") {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.requestDefaults.context1m must be a boolean",
- path: ["requestDefaults", "context1m"],
- });
- }
- }
- }
-
- // [Oracle CONDITIONAL] consoleApiKey는 bailian-coding-plan 전용 필드.
- // 다른 프로바이더 공통 규약으로 재사용하지 않는다.
- const consoleApiKey = data.consoleApiKey;
- if (consoleApiKey !== undefined && consoleApiKey !== null && typeof consoleApiKey !== "string") {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.consoleApiKey must be a string",
- path: ["consoleApiKey"],
- });
- }
- if (typeof consoleApiKey === "string" && consoleApiKey.length > 10000) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.consoleApiKey must be at most 10000 characters",
- path: ["consoleApiKey"],
- });
- }
-
- const groupTag = data.tag;
- if (
- groupTag !== undefined &&
- groupTag !== null &&
- (typeof groupTag !== "string" || groupTag.length > 100)
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.tag must be a string up to 100 chars",
- path: ["tag"],
- });
- }
-
- const routingTags = data.tags;
- if (routingTags !== undefined && routingTags !== null) {
- if (!Array.isArray(routingTags) || routingTags.length > 50) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.tags must be an array with at most 50 items",
- path: ["tags"],
- });
- } else if (
- routingTags.some(
- (tag) => typeof tag !== "string" || tag.trim().length === 0 || tag.trim().length > 64
- )
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message:
- "providerSpecificData.tags must contain non-empty strings up to 64 characters each",
- path: ["tags"],
- });
- }
- }
-
- const excludedModels = data.excludedModels ?? data.excluded_models;
- if (excludedModels !== undefined && excludedModels !== null) {
- if (typeof excludedModels === "string") {
- if (excludedModels.length > 5000) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.excludedModels string must be up to 5000 chars",
- path: ["excludedModels"],
- });
- }
- } else if (!Array.isArray(excludedModels) || excludedModels.length > 100) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message: "providerSpecificData.excludedModels must be an array with at most 100 items",
- path: ["excludedModels"],
- });
- } else if (
- excludedModels.some(
- (pattern) =>
- typeof pattern !== "string" ||
- pattern.trim().length === 0 ||
- pattern.trim().length > 200 ||
- pattern.trim() === "**"
- )
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message:
- "providerSpecificData.excludedModels must contain non-empty patterns up to 200 characters",
- path: ["excludedModels"],
- });
- }
- }
-
- const clientProfile = data.clientProfile;
- if (clientProfile !== undefined && clientProfile !== null) {
- const normalized = typeof clientProfile === "string" ? clientProfile.trim().toLowerCase() : "";
- if (
- typeof clientProfile !== "string" ||
- !["ide", "harness", "cli", "sdk"].includes(normalized)
- ) {
- ctx.addIssue({
- code: z.ZodIssueCode.custom,
- message:
- "providerSpecificData.clientProfile must be ide, harness, cli, or sdk (cli/sdk map to harness)",
- path: ["clientProfile"],
- });
- }
- }
-}
+import { validateProviderSpecificData } from "./providerSpecificData";
// Re-export validation helpers from dedicated module to avoid webpack barrel-file
// optimization bug that truncates exports from large files.
diff --git a/tests/unit/build/check-tracked-artifacts.test.ts b/tests/unit/build/check-tracked-artifacts.test.ts
index f3edfc83fdb..99d8c05960b 100644
--- a/tests/unit/build/check-tracked-artifacts.test.ts
+++ b/tests/unit/build/check-tracked-artifacts.test.ts
@@ -30,6 +30,11 @@ test("checkTrackedArtifacts: quality-metrics.json is flagged", () => {
assert.equal(result.length, 1);
});
+test("checkTrackedArtifacts: config/quality/quality-metrics.json is flagged", () => {
+ const result = checkTrackedArtifacts(["config/quality/quality-metrics.json"]);
+ assert.equal(result.length, 1);
+});
+
test("checkTrackedArtifacts: symlink mode (120000) is flagged", () => {
const result = checkTrackedArtifacts([], ["node_modules"]);
assert.equal(result.length, 1);
diff --git a/tests/unit/check-deps.test.ts b/tests/unit/check-deps.test.ts
index 864493efcc0..6adaf44c483 100644
--- a/tests/unit/check-deps.test.ts
+++ b/tests/unit/check-deps.test.ts
@@ -18,17 +18,13 @@ test("no unapproved deps when all are allowlisted", () => {
});
test("flags a dependency not on the allowlist (potential slopsquat)", () => {
- assert.deepEqual(
- findUnapprovedDeps(["react", "reactt-router"], new Set(["react"])),
- ["reactt-router"]
- );
+ assert.deepEqual(findUnapprovedDeps(["react", "reactt-router"], new Set(["react"])), [
+ "reactt-router",
+ ]);
});
test("flags multiple new deps, preserves order, de-dupes", () => {
- assert.deepEqual(
- findUnapprovedDeps(["a", "b", "a", "c"], new Set(["a"])),
- ["b", "c"]
- );
+ assert.deepEqual(findUnapprovedDeps(["a", "b", "a", "c"], new Set(["a"])), ["b", "c"]);
});
// --- 6A.8: automatic workspace discovery ---
@@ -64,7 +60,7 @@ test("6A.8: discoverManifests does NOT include node_modules, .next, or deep refe
});
test("6A.8: all workspace package deps are in the allowlist (gate exits 0 with expanded scope)", () => {
- const allowlistPath = path.join(repoRoot, "dependency-allowlist.json");
+ const allowlistPath = path.join(repoRoot, "config/quality/dependency-allowlist.json");
const allowlist = new Set(JSON.parse(fs.readFileSync(allowlistPath, "utf8")).allowed || []);
const manifests = discoverManifests(repoRoot);
const allDeps: string[] = [];
@@ -80,7 +76,11 @@ test("6A.8: all workspace package deps are in the allowlist (gate exits 0 with e
);
}
const unapproved = findUnapprovedDeps(allDeps, allowlist);
- assert.deepEqual(unapproved, [], `expected all deps to be approved, got: ${unapproved.join(", ")}`);
+ assert.deepEqual(
+ unapproved,
+ [],
+ `expected all deps to be approved, got: ${unapproved.join(", ")}`
+ );
});
// --- 6A.8: stale-allowlist enforcement ---
@@ -102,10 +102,9 @@ test("6A.8 stale: a dep removed from all manifests is detected as stale in allow
test("7.8 evaluateDepAge: package older than 72h is OK", () => {
const now = Date.now();
const createdMs = now - 73 * 60 * 60 * 1000; // 73 hours ago
- const { ok, ageHours } = (evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number })(
- createdMs,
- now
- );
+ const { ok, ageHours } = (
+ evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number }
+ )(createdMs, now);
assert.ok(ok, "package older than 72h should be ok");
assert.ok(ageHours >= 73, "ageHours should reflect elapsed time");
});
@@ -113,20 +112,18 @@ test("7.8 evaluateDepAge: package older than 72h is OK", () => {
test("7.8 evaluateDepAge: package exactly at 72h boundary is OK (boundary inclusive)", () => {
const now = Date.now();
const createdMs = now - 72 * 60 * 60 * 1000; // exactly 72 hours ago
- const { ok } = (evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number })(
- createdMs,
- now
- );
+ const { ok } = (
+ evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number }
+ )(createdMs, now);
assert.ok(ok, "package at exactly 72h should be ok (inclusive boundary)");
});
test("7.8 evaluateDepAge: package published 1h ago is NOT OK", () => {
const now = Date.now();
const createdMs = now - 1 * 60 * 60 * 1000; // 1 hour ago
- const { ok, ageHours } = (evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number })(
- createdMs,
- now
- );
+ const { ok, ageHours } = (
+ evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number }
+ )(createdMs, now);
assert.ok(!ok, "package published 1h ago should NOT be ok");
assert.ok(ageHours < 72, "ageHours should reflect < 72h");
});
@@ -135,28 +132,23 @@ test("7.8 evaluateDepAge: respects custom minAgeHours", () => {
const now = Date.now();
const createdMs = now - 10 * 60 * 60 * 1000; // 10 hours ago
// With 24h minimum, 10h-old package should fail
- const { ok: failWith24 } = (evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number })(
- createdMs,
- now,
- 24
- );
+ const { ok: failWith24 } = (
+ evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number }
+ )(createdMs, now, 24);
assert.ok(!failWith24, "10h-old package should fail with 24h minimum");
// With 6h minimum, 10h-old package should pass
- const { ok: passWith6 } = (evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number })(
- createdMs,
- now,
- 6
- );
+ const { ok: passWith6 } = (
+ evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number }
+ )(createdMs, now, 6);
assert.ok(passWith6, "10h-old package should pass with 6h minimum");
});
test("7.8 evaluateDepAge: future timestamp (time.created in future) is NOT OK", () => {
const now = Date.now();
const createdMs = now + 1000; // 1s in future (clock skew edge case)
- const { ok, ageHours } = (evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number })(
- createdMs,
- now
- );
+ const { ok, ageHours } = (
+ evaluateDepAge as (a: number, b: number, c?: number) => { ok: boolean; ageHours: number }
+ )(createdMs, now);
assert.ok(!ok, "future timestamp should not be ok");
assert.ok(ageHours < 0, "ageHours should be negative for future timestamp");
});
@@ -169,11 +161,17 @@ test("7.8 evaluateDepAge: future timestamp (time.created in future) is NOT OK",
// - any real npm package like "react" is found and old → does NOT appear in any list
test("7.8 auditNewDepsRegistry: empty dep list returns all-empty results", () => {
- const result = (auditNewDepsRegistry as (deps: string[], minAge?: number, now?: number) => {
- notFound: string[];
- tooNew: Array<{ name: string; ageHours: number }>;
- offline: string[];
- })([], 72, Date.now());
+ const result = (
+ auditNewDepsRegistry as (
+ deps: string[],
+ minAge?: number,
+ now?: number
+ ) => {
+ notFound: string[];
+ tooNew: Array<{ name: string; ageHours: number }>;
+ offline: string[];
+ }
+ )([], 72, Date.now());
assert.deepEqual(result.notFound, []);
assert.deepEqual(result.tooNew, []);
assert.deepEqual(result.offline, []);
diff --git a/tests/unit/combo-auto-candidate-expansion.test.ts b/tests/unit/combo-auto-candidate-expansion.test.ts
index 61d51200f2f..0d54d630485 100644
--- a/tests/unit/combo-auto-candidate-expansion.test.ts
+++ b/tests/unit/combo-auto-candidate-expansion.test.ts
@@ -92,6 +92,29 @@ test("expandAutoComboCandidatePool is a no-op when an explicit candidatePool exi
assert.equal(result[0].modelStr, "openai/gpt-4o");
});
+test("expandAutoComboCandidatePool falls through to active connections when candidatePool is an empty array", async () => {
+ await providersDb.createProviderConnection({
+ provider: "openai",
+ authType: "apikey",
+ name: "OpenAI",
+ apiKey: "sk-test-openai",
+ defaultModel: "gpt-4o-mini",
+ });
+
+ const expanded = await combo.expandAutoComboCandidatePool([], {
+ config: { auto: { candidatePool: [] } },
+ });
+
+ // An empty candidatePool should NOT trigger early return — the function
+ // should fall through and expand from active connections instead.
+ assert.ok(
+ expanded.length > 0,
+ "expected expansion from active connections despite empty candidatePool"
+ );
+ const openaiTargets = expanded.filter((t) => t.provider === "openai");
+ assert.ok(openaiTargets.length > 0, "expected openai targets to be expanded");
+});
+
test("expandAutoComboCandidatePool does not duplicate an already-present modelStr", async () => {
await providersDb.createProviderConnection({
provider: "openai",
diff --git a/tests/unit/combo-strategy-fallbacks.test.ts b/tests/unit/combo-strategy-fallbacks.test.ts
index f6e38ed34c4..f8ab628faaa 100644
--- a/tests/unit/combo-strategy-fallbacks.test.ts
+++ b/tests/unit/combo-strategy-fallbacks.test.ts
@@ -18,7 +18,8 @@ const settingsDb = await import("../../src/lib/db/settings.ts");
const { resetAllComboMetrics } = await import("../../open-sse/services/comboMetrics.ts");
const { resetAllCircuitBreakers, getCircuitBreaker } =
await import("../../src/shared/utils/circuitBreaker.ts");
-const { resetAll: resetAllSemaphores } = await import("../../open-sse/services/rateLimitSemaphore.ts");
+const { resetAll: resetAllSemaphores } =
+ await import("../../open-sse/services/rateLimitSemaphore.ts");
const { _resetAllDecks } = await import("../../src/shared/utils/shuffleDeck.ts");
const { clearSessions } = await import("../../open-sse/services/sessionManager.ts");
@@ -290,6 +291,74 @@ test("strict-random falls back to the remaining target when the deck pick fails"
assert.notEqual(calls[0], calls[1]);
});
+test("round-robin uses existing stickyRoundRobinLimit for combo target batching", async () => {
+ const calls: string[] = [];
+ const combo = {
+ name: "rr-sticky-combo-batches",
+ strategy: "round-robin",
+ models: ["openai/a", "claude/b", "gemini/c"],
+ config: { maxRetries: 0, retryDelayMs: 0, fallbackDelayMs: 0 },
+ };
+
+ for (let i = 0; i < 10; i += 1) {
+ const result = await handleComboChat({
+ body: {},
+ combo,
+ handleSingleModel: async (_body: any, modelStr: string) => {
+ calls.push(modelStr);
+ return okResponse();
+ },
+ isModelAvailable: async () => true,
+ log: createLog(),
+ settings: { stickyRoundRobinLimit: 3 },
+ allCombos: null,
+ });
+ assert.equal(result.ok, true);
+ }
+
+ assert.deepEqual(calls, [
+ "openai/a",
+ "openai/a",
+ "openai/a",
+ "claude/b",
+ "claude/b",
+ "claude/b",
+ "gemini/c",
+ "gemini/c",
+ "gemini/c",
+ "openai/a",
+ ]);
+});
+
+test("round-robin sticky batching fallback success becomes sticky target", async () => {
+ const calls: string[] = [];
+ const combo = {
+ name: "rr-sticky-fallback-success",
+ strategy: "round-robin",
+ models: ["openai/a", "claude/b", "gemini/c"],
+ config: { maxRetries: 0, retryDelayMs: 0, fallbackDelayMs: 0 },
+ };
+
+ for (let i = 0; i < 4; i += 1) {
+ const result = await handleComboChat({
+ body: {},
+ combo,
+ handleSingleModel: async (_body: any, modelStr: string) => {
+ calls.push(modelStr);
+ if (modelStr === "openai/a") return errorResponse(503, "a is down");
+ return okResponse();
+ },
+ isModelAvailable: async () => true,
+ log: createLog(),
+ settings: { stickyRoundRobinLimit: 2 },
+ allCombos: null,
+ });
+ assert.equal(result.ok, true);
+ }
+
+ assert.deepEqual(calls, ["openai/a", "claude/b", "claude/b", "gemini/c", "gemini/c"]);
+});
+
test("strict-random survives a stale deck entry after a target is removed", async () => {
const comboTwoTargets = {
name: "strict-random-stale",
diff --git a/tests/unit/executor-default-base.test.ts b/tests/unit/executor-default-base.test.ts
index 1cbe3e0e867..45aa954cafc 100644
--- a/tests/unit/executor-default-base.test.ts
+++ b/tests/unit/executor-default-base.test.ts
@@ -103,10 +103,7 @@ test("DefaultExecutor.buildUrl uses full chat endpoints for hosted OpenAI-compat
bazaarlink.buildUrl("auto:free", true),
"https://bazaarlink.ai/api/v1/chat/completions"
);
- assert.equal(
- crof.buildUrl("gpt-4.1", true),
- "https://crof.ai/v1/chat/completions"
- );
+ assert.equal(crof.buildUrl("gpt-4.1", true), "https://crof.ai/v1/chat/completions");
});
test("DefaultExecutor.buildUrl handles openai-compatible and anthropic-compatible providers", () => {
@@ -742,6 +739,44 @@ test("DefaultExecutor.transformRequest respects disableStreamOptions for OpenAI
assert.deepEqual((chatResultEnabled as any).stream_options, { include_usage: true });
});
+test("DefaultExecutor.transformRequest injects OpenRouter connection preset", () => {
+ const executor = new DefaultExecutor("openrouter");
+ const body = { model: "openai/gpt-4", messages: [{ role: "user", content: "hi" }] };
+
+ const result = executor.transformRequest("openai/gpt-4", body, true, {
+ providerSpecificData: { preset: " email-copywriter " },
+ });
+
+ assert.equal((result as any).preset, "email-copywriter");
+ assert.deepEqual((result as any).stream_options, { include_usage: true });
+ assert.equal((body as any).preset, undefined);
+
+ const explicit = executor.transformRequest(
+ "openai/gpt-4",
+ { ...body, preset: "client-preset" },
+ true,
+ { providerSpecificData: { preset: "connection-preset" } }
+ );
+
+ assert.equal((explicit as any).preset, "client-preset");
+
+ const explicitNull = executor.transformRequest("openai/gpt-4", { ...body, preset: null }, true, {
+ providerSpecificData: { preset: "connection-preset" },
+ });
+ assert.equal((explicitNull as any).preset, null);
+
+ const explicitEmpty = executor.transformRequest("openai/gpt-4", { ...body, preset: "" }, true, {
+ providerSpecificData: { preset: "connection-preset" },
+ });
+ assert.equal((explicitEmpty as any).preset, "");
+
+ const blank = executor.transformRequest("openai/gpt-4", body, true, {
+ providerSpecificData: { preset: " " },
+ });
+
+ assert.equal((blank as any).preset, undefined);
+});
+
test("DefaultExecutor.transformRequest strips stream_options from Anthropic-compatible targets", () => {
const anthropicCompat = new DefaultExecutor("anthropic-compatible-test");
const anthropicCcCompat = new DefaultExecutor("anthropic-compatible-cc-test");
diff --git a/tests/unit/provider-models-config.test.ts b/tests/unit/provider-models-config.test.ts
index 403d32e6cf8..d95b50c1ac5 100644
--- a/tests/unit/provider-models-config.test.ts
+++ b/tests/unit/provider-models-config.test.ts
@@ -53,6 +53,25 @@ test("provider models helpers resolve provider IDs through aliases", () => {
assert.deepEqual(getModelsByProviderId("provider-that-does-not-exist"), []);
});
+test("getProviderModels returns models for both the alias and the raw provider id", () => {
+ // Pick a provider whose alias differs from its id (e.g. "github" → "gh").
+ const aliased = Object.entries(PROVIDER_ID_TO_ALIAS).find(([id, a]) => id !== a) as
+ | [string, string]
+ | undefined;
+ if (!aliased) return; // no aliased providers → trivially satisfied
+
+ const [rawId, alias] = aliased;
+ const byAlias = getProviderModels(alias);
+ const byRawId = getProviderModels(rawId);
+
+ assert.ok(byAlias.length > 0, `expected models under alias "${alias}"`);
+ assert.deepEqual(
+ byRawId,
+ byAlias,
+ `getProviderModels("${rawId}") should return the same models as getProviderModels("${alias}")`
+ );
+});
+
test("Reka registry exposes preset models", () => {
const rekaModels = getModelsByProviderId("reka");
const ids = rekaModels.map((model) => model.id);
diff --git a/tests/unit/provider-specific-data-schema.test.ts b/tests/unit/provider-specific-data-schema.test.ts
index e8dbfa64e12..08f76391eee 100644
--- a/tests/unit/provider-specific-data-schema.test.ts
+++ b/tests/unit/provider-specific-data-schema.test.ts
@@ -127,3 +127,47 @@ test("provider schemas reject unknown Codex service tiers", () => {
assert.equal(created.success, false);
assert.equal(updated.success, false);
});
+
+test("provider schemas accept OpenRouter preset in providerSpecificData", () => {
+ const created = createProviderSchema.safeParse({
+ provider: "openrouter",
+ apiKey: "token",
+ name: "OpenRouter",
+ providerSpecificData: {
+ preset: "email-copywriter",
+ },
+ });
+ const updated = updateProviderConnectionSchema.safeParse({
+ providerSpecificData: {
+ preset: "code-reviewer",
+ },
+ });
+ const padded = updateProviderConnectionSchema.safeParse({
+ providerSpecificData: {
+ preset: `${" ".repeat(120)}prefer${" ".repeat(120)}`,
+ },
+ });
+
+ assert.equal(created.success, true);
+ assert.equal(updated.success, true);
+ assert.equal(padded.success, true);
+});
+
+test("provider schemas reject oversized OpenRouter preset values", () => {
+ const created = createProviderSchema.safeParse({
+ provider: "openrouter",
+ apiKey: "token",
+ name: "OpenRouter",
+ providerSpecificData: {
+ preset: "x".repeat(201),
+ },
+ });
+ const updated = updateProviderConnectionSchema.safeParse({
+ providerSpecificData: {
+ preset: 123,
+ },
+ });
+
+ assert.equal(created.success, false);
+ assert.equal(updated.success, false);
+});
diff --git a/tests/unit/request-defaults-store-session.test.ts b/tests/unit/request-defaults-store-session.test.ts
index 1aa04506aaa..64057ba66b4 100644
--- a/tests/unit/request-defaults-store-session.test.ts
+++ b/tests/unit/request-defaults-store-session.test.ts
@@ -70,3 +70,37 @@ test("normalizeProviderSpecificData keeps only boolean CC-compatible 1M request
customFlag: "keep-me",
});
});
+
+test("normalizeProviderSpecificData trims OpenRouter preset and clears empty values", () => {
+ const normalized = normalizeProviderSpecificData("openrouter", {
+ preset: " email-copywriter ",
+ tag: "primary",
+ });
+
+ assert.equal(normalized?.preset, "email-copywriter");
+ assert.equal(normalized?.tag, "primary");
+
+ const stripped = normalizeProviderSpecificData("openrouter", {
+ preset: " ",
+ tag: "primary",
+ });
+
+ assert.equal(stripped?.preset, undefined);
+ assert.equal(stripped?.tag, "primary");
+
+ const oversized = normalizeProviderSpecificData("openrouter", {
+ preset: "x".repeat(201),
+ tag: "primary",
+ });
+
+ assert.equal(oversized?.preset, undefined);
+ assert.equal(oversized?.tag, "primary");
+
+ const ignored = normalizeProviderSpecificData("openai", {
+ preset: "email-copywriter",
+ tag: "primary",
+ });
+
+ assert.equal(ignored?.preset, undefined);
+ assert.equal(ignored?.tag, "primary");
+});
diff --git a/tests/unit/sync-wiki.test.ts b/tests/unit/sync-wiki.test.ts
new file mode 100644
index 00000000000..33e35cfe8c2
--- /dev/null
+++ b/tests/unit/sync-wiki.test.ts
@@ -0,0 +1,71 @@
+// Unit tests for the pure helpers in scripts/docs/sync-wiki.mjs (the GitHub wiki sync).
+import { test } from "node:test";
+import assert from "node:assert/strict";
+
+import {
+ normKey,
+ toWikiName,
+ toWikiContent,
+ syncHomeCounts,
+ parseWikiPage,
+ WIKI_BANNER,
+ NEW_PAGE_EXCLUDE,
+} from "../../scripts/docs/sync-wiki.mjs";
+
+test("normKey: case/separator-insensitive fuzzy key (matches docs basename to curated wiki name)", () => {
+ // The wiki names are hand-curated and not deterministic — normKey is what lets us
+ // match e.g. FLY_IO_DEPLOYMENT_GUIDE.md to the existing "Fly-io-Deployment-Guide" page.
+ assert.equal(normKey("FLY_IO_DEPLOYMENT_GUIDE.md"), normKey("Fly-io-Deployment-Guide"));
+ assert.equal(normKey("API_REFERENCE"), normKey("API-Reference"));
+ assert.equal(normKey("A2A-SERVER.md"), "a2aserver");
+});
+
+test("toWikiName: acronym-aware Title-Case-dashed name for NEW pages", () => {
+ assert.equal(toWikiName("SUPPLY_CHAIN.md"), "Supply-Chain");
+ assert.equal(toWikiName("API_REFERENCE"), "API-Reference");
+ assert.equal(toWikiName("QUOTA_SHARE"), "Quota-Share");
+ assert.equal(toWikiName("ACP"), "ACP");
+});
+
+test("toWikiContent: strips YAML frontmatter and prepends the language banner", () => {
+ const doc = '---\ntitle: "X"\nversion: 3.8.2\n---\n\n# Heading\n\nBody text.\n';
+ const out = toWikiContent(doc);
+ assert.ok(out.startsWith(WIKI_BANNER), "must start with the wiki language banner");
+ assert.ok(!out.includes("version: 3.8.2"), "frontmatter must be removed");
+ assert.ok(out.includes("# Heading"), "content body must be preserved");
+ assert.ok(out.endsWith("Body text.\n"), "trailing newline normalized");
+});
+
+test("toWikiContent: a document without frontmatter is kept intact (plus banner)", () => {
+ const doc = "# No Frontmatter\n\nHello.\n";
+ const out = toWikiContent(doc);
+ assert.equal(out, WIKI_BANNER + "# No Frontmatter\n\nHello.\n");
+});
+
+test("syncHomeCounts: rewrites the cover-page provider/strategy counts", () => {
+ const home = "Connect every AI tool to 177 providers.\n**177 AI Providers** · **14 Routing Strategies**\n";
+ const out = syncHomeCounts(home, { providers: 226, strategies: 15, mcpTools: null, locales: 42 });
+ assert.ok(out.includes("226 providers"));
+ assert.ok(out.includes("**226 AI Providers**"));
+ assert.ok(out.includes("**15 Routing Strategies**"));
+ assert.ok(!out.includes("177"));
+});
+
+test("syncHomeCounts: leaves text untouched when a count is null (best-effort MCP)", () => {
+ const home = "| **MCP Server** | 87 tools |\n";
+ const out = syncHomeCounts(home, { providers: null, strategies: null, mcpTools: null, locales: null });
+ assert.equal(out, home);
+});
+
+test("parseWikiPage: splits the locale prefix (U+2010) from the page name", () => {
+ assert.deepEqual(parseWikiPage("Architecture"), { locale: null, name: "Architecture" });
+ // U+2010 HYPHEN separator used by the localized mirrors (e.g. "pt-BR‐Architecture").
+ assert.deepEqual(parseWikiPage("pt-BR‐Architecture"), { locale: "pt-BR", name: "Architecture" });
+ assert.deepEqual(parseWikiPage("phi‐API-Reference"), { locale: "phi", name: "API-Reference" });
+});
+
+test("NEW_PAGE_EXCLUDE: internal reports/plans never become public wiki pages", () => {
+ assert.ok(NEW_PAGE_EXCLUDE.has("DOCUMENTATION_AUDIT_REPORT"));
+ assert.ok(NEW_PAGE_EXCLUDE.has("README"));
+ assert.ok(!NEW_PAGE_EXCLUDE.has("SUPPLY_CHAIN"));
+});