diff --git a/.github/workflows/wiki-sync.yml b/.github/workflows/wiki-sync.yml new file mode 100644 index 00000000000..7381ed56483 --- /dev/null +++ b/.github/workflows/wiki-sync.yml @@ -0,0 +1,69 @@ +name: Wiki Sync + +# Keeps the GitHub wiki in sync with docs/ on every release that lands on main. +# The wiki has no native generator and historically drifts (it sat at "212+ providers / +# 14 strategies / 37 MCP tools" while code was at 226 / 15 / 87, and new docs like +# SUPPLY_CHAIN never appeared). This runs scripts/docs/sync-wiki.mjs, which: +# - ADDS any docs/ page missing from the wiki (curated; internal reports excluded), +# - syncs the four cover-page counts on Home.md. +# It does NOT overwrite existing wiki pages by default: several docs sources still carry +# stale counts (e.g. ARCHITECTURE.md says "177 providers" while the wiki cover is 226), +# so blind overwrite would regress the wiki. Full content parity (--update-existing) is +# gated on regenerating those sources — see docs/ops/DOCUMENTATION_AUDIT_REPORT.md. + +on: + push: + branches: [main] + paths: + - "docs/**" + - "README.md" + - "AGENTS.md" + - "src/shared/constants/routingStrategies.ts" + - "config/i18n.json" + - "open-sse/mcp-server/server.ts" + - "scripts/docs/sync-wiki.mjs" + workflow_dispatch: + +permissions: + contents: write + +concurrency: + group: wiki-sync + cancel-in-progress: false + +jobs: + sync-wiki: + name: Sync wiki with docs + runs-on: ubuntu-latest + steps: + - name: Checkout repo + uses: actions/checkout@v4 + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: "24" + + - name: Clone wiki + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REPO: ${{ github.repository }} + run: | + git clone "https://x-access-token:${GH_TOKEN}@github.com/${REPO}.wiki.git" wiki + + - name: Sync wiki (add missing pages + cover counts) + run: node scripts/docs/sync-wiki.mjs --wiki-dir wiki + + - name: Commit & push if changed + run: | + cd wiki + if [ -n "$(git status --porcelain)" ]; then + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add -A + git commit -m "docs(wiki): auto-sync pages + cover counts with docs" + git push + echo "Wiki updated." + else + echo "Wiki already in sync — nothing to push." + fi diff --git a/.gitignore b/.gitignore index cbe3af7f64d..1a94d63b2b7 100644 --- a/.gitignore +++ b/.gitignore @@ -203,8 +203,11 @@ pr_reviews*.json # internal setup prompts with personal credentials — never commit CODEX-SETUP-PROMPT.md -# Quality ratchet — métricas efêmeras (baseline é commitado, métricas não) -quality-metrics.json +# Quality ratchet — métricas efêmeras (baseline commitado em config/quality/; métricas não) +config/quality/quality-metrics.json + +# Runtime logs (diretório local, nunca versionado) +/logs/ -home-diegosouzapw-dev-automações-bots-yt-downloader-20260504 .txt -home-diegosouzapw-dev-automações-bots-yt-downloader-20260410 .txt docs/prompts/AGENT-OWNERSHIP-PROTOCOL.omniroute.md @@ -212,3 +215,4 @@ docs/prompts/AGENT-OWNERSHIP-PROTOCOL.md docs/prompts/AGENT-OWNERSHIP-PROTOCOL.omniroute-mim.md docs/prompts/AGENT-OWNERSHIP-PROTOCOL.omniroute-mid.md omniroute.md +quality-metrics.json diff --git a/CHANGELOG.md b/CHANGELOG.md index bcdcd4e4b5a..b6aa2eb465b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,15 @@ --- +## [3.8.26] — TBD + +### 🧹 Internal / Quality / Docs + +- **fix(ci): grant `contents: write` to the npm publish job for SBOM attach** — the v3.8.25 TokenPermissions hardening set the npm-publish `publish` job to `contents: read`, but its "Attach SBOM to GitHub Release" step (`gh release upload`) needs `contents: write` and failed with HTTP 403 on the v3.8.25 release (npm / GitHub Packages / opencode-plugin / Docker / Electron all published fine; only the SBOM attach broke — the v3.8.25 SBOM was attached manually). ([#3874](https://github.com/diegosouzapw/OmniRoute/pull/3874) — thanks @diegosouzapw) +- **docs: refresh the provider count to 226 + regenerate `PROVIDER_REFERENCE.md`** — the README advertised a stale `177 providers`; the canonical generator (`scripts/docs/gen-provider-reference.ts`) now reports **226 unique provider IDs**, so the README badges/anchors and the generated provider reference were brought in sync. Also adds a documentation audit/sync report. (thanks @diegosouzapw) + +--- + ## [3.8.25] — 2026-06-14 ### ✨ New Features diff --git a/README.md b/README.md index 950f1b27302..b2783bea126 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ # 🚀 OmniRoute — The Free AI Gateway -### Never stop coding. Connect every AI tool to **177 providers** — **50+ free** — through one endpoint. +### Never stop coding. Connect every AI tool to **226 providers** — **50+ free** — through one endpoint. **Plug Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini. Auto-fallback.**
@@ -19,8 +19,8 @@
-[![177 AI Providers](https://img.shields.io/badge/177-AI_Providers-6C5CE7?style=for-the-badge)](#-177-ai-providers--50-free) -[![50+ Free](https://img.shields.io/badge/50%2B-Free_Tiers-00B894?style=for-the-badge)](#-177-ai-providers--50-free) +[![226 AI Providers](https://img.shields.io/badge/226-AI_Providers-6C5CE7?style=for-the-badge)](#-226-ai-providers--50-free) +[![50+ Free](https://img.shields.io/badge/50%2B-Free_Tiers-00B894?style=for-the-badge)](#-226-ai-providers--50-free) [![1.9B+ Free Tokens/mo](https://img.shields.io/badge/1.9B%2B-Free_Tokens%2Fmo-00B894?style=for-the-badge)](docs/reference/FREE_TIERS.md) [![Token Savings](https://img.shields.io/badge/up_to_95%25-Token_Savings-E17055?style=for-the-badge)](#%EF%B8%8F-save-1595-tokens--automatically) [![15 Strategies](https://img.shields.io/badge/15-Routing_Strategies-0984E3?style=for-the-badge)](#-combos--the-flagship) @@ -59,7 +59,7 @@
-[**🚀 Quick Start**](#-quick-start) • [**🎯 Combos**](#-combos--the-flagship) • [**🌐 Providers**](#-177-ai-providers--50-free) • [**🔌 CLI & MCP**](#-full-cli--a2a--mcp) • [**🗜️ Compression**](#%EF%B8%8F-save-1595-tokens--automatically) • [**🌍 Website**](https://omniroute.online) +[**🚀 Quick Start**](#-quick-start) • [**🎯 Combos**](#-combos--the-flagship) • [**🌐 Providers**](#-226-ai-providers--50-free) • [**🔌 CLI & MCP**](#-full-cli--a2a--mcp) • [**🗜️ Compression**](#%EF%B8%8F-save-1595-tokens--automatically) • [**🌍 Website**](https://omniroute.online) [💥 The Promise](#-the-promise) • [🤔 Why](#-why-omniroute) • [🏆 What Sets Apart](#-what-sets-omniroute-apart) • [🤖 Compatible CLIs](#-compatible-clis--coding-agents) • [🖥️ Where It Runs](#%EF%B8%8F-where-omniroute-runs--anywhere) • [🔒 Private](#-private--local-first) • [🎬 In Action](#-omniroute-in-action) • [📚 Explore More](#-explore-more) • [📧 Support](#-support--community) @@ -136,18 +136,18 @@ -> One endpoint. **177 providers.** Never stop building — and let OmniRoute pick the cheapest one that works. +> One endpoint. **226 providers.** Never stop building — and let OmniRoute pick the cheapest one that works. - + - +
🚫 Never hit limits
Auto-fallback across 177 providers in milliseconds. Quota out? Next provider takes over — zero downtime.
🚫 Never hit limits
Auto-fallback across 226 providers in milliseconds. Quota out? Next provider takes over — zero downtime.
💸 Save up to 95% tokens
RTK + Caveman stacked compression cuts 15–95% of eligible tokens (~89% avg on tool-heavy sessions).
🆓 $0 to start
50+ providers with a free tier, 11 free forever (Kiro, Qoder, Pollinations, LongCat…). No card needed.
🔌 Every tool works
16+ coding agents — Claude Code, Codex, Cursor, Cline, Copilot, Antigravity — through one config.
🧩 One endpoint
OpenAI ↔ Claude ↔ Gemini ↔ Responses API translation. Point any tool at /v1 and it just works.
🛡️ Production-grade
Circuit breakers, TLS stealth, MCP (87 tools), A2A, memory, guardrails, evals. 4,690+ tests.
🛡️ Production-grade
Circuit breakers, TLS stealth, MCP (87 tools), A2A, memory, guardrails, evals. 14,965 tests.
@@ -263,7 +263,7 @@ Result: 4 layers of fallback = zero downtime | Feature | OmniRoute | Other routers | | -------------------------------------- | ----------------------------------------------------------- | ------------- | -| 🌐 Providers | **177** | 20–100 | +| 🌐 Providers | **226** | 20–100 | | 🆓 Free providers | **50+ (11 free forever)** | 1–5 | | 🔀 Routing strategies | **15** (priority, weighted, cost-optimized, context-relay…) | 1–3 | | 🗜️ Token compression | **RTK + Caveman stacked (15–95%)** | None / 20–40% | @@ -274,7 +274,7 @@ Result: 4 layers of fallback = zero downtime | ☁️ Cloud agents | **Codex, Devin, Jules** | None | | 🥷 TLS fingerprint stealth | **JA3/JA4 via wreq-js** | None | | 🖥️ Multi-platform | **Web · Desktop · Termux · PWA** | Web only | -| 🌍 i18n | **40+ locales** | 0–4 | +| 🌍 i18n | **42 locales** | 0–4 | 📊 Detailed comparison vs LiteLLM, OpenRouter & Portkey → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md) @@ -319,11 +319,11 @@ Result: 4 layers of fallback = zero downtime
-# 🌐 177 AI Providers — 50+ Free +# 🌐 226 AI Providers — 50+ Free
-> The most complete catalog of any open-source router: **177 providers**, **50+ with a free tier**, **11 free forever**. +> The most complete catalog of any open-source router: **226 providers**, **50+ with a free tier**, **11 free forever**.
@@ -439,7 +439,25 @@ claude mcp add-server omniroute --type http --url http://localhost:20128/api/mcp
-> **Why use many token when few token do trick?** Every request passes through OmniRoute's compression pipeline **transparently** — no client changes. It stacks ideas from [RTK](https://github.com/rtk-ai/rtk), [Caveman](https://github.com/JuliusBrussee/caveman) (⭐ 51K+), and [Troglodita](https://github.com/leninejunior/troglodita) (PT-BR). +> **Why use many token when few token do trick?** Every request passes through OmniRoute's compression pipeline **transparently** — no client changes. It's now a **stack of 9 composable engines** that run in order and mix & match per routing combo — building on ideas from [RTK](https://github.com/rtk-ai/rtk), [Caveman](https://github.com/JuliusBrussee/caveman) (⭐ 51K+), [LLMLingua-2](https://github.com/microsoft/LLMLingua), and [Troglodita](https://github.com/leninejunior/troglodita) (PT-BR). + +### 🧱 The 9-engine stack + +Engines run in pipeline order; each is independently toggleable and configurable per combo: + +| # | Engine | What it does | +| --- | ----------------- | ------------------------------------------------------------------------ | +| 1 | **Session-Dedup** | Drops content repeated across turns (content-addressed, cross-turn) | +| 2 | **CCR** | Archives large blocks behind retrieve markers, fetched on demand | +| 3 | **RTK** | Smart tool-result filtering, dedup & truncation (command-aware) | +| 4 | **Headroom** | Lossless tabular compaction of homogeneous JSON arrays (~30%+) | +| 5 | **Caveman** | Rule-based prose compression (~65–75% on output) | +| 6 | **LLMLingua-2** | ML semantic pruning via MobileBERT ONNX — code-safe, async | +| 7 | **Lite** | Whitespace + image-URL trimming (latency-light baseline) | +| 8 | **Aggressive** | Summarization + progressive aging of old turns | +| 9 | **Ultra** | Heuristic token pruning with an optional small-model (SLM) tier | + +Code blocks, URLs and structured data are **always preserved** byte-perfect. **One-click presets** combine the engines: | Mode | Savings | Best for | | ------------------------------ | ---------- | --------------------------- | @@ -471,7 +489,7 @@ claude mcp add-server omniroute --type http --url http://localhost:20128/api/mcp ### 📖 How it works — pipeline, architecture & savings math ``` -Client (10,000 tok) ──▶ OmniRoute Compression (7 options) ──▶ Provider (~1,080 tok, up to 95% saved) +Client (10,000 tok) ──▶ OmniRoute Compression (9 engines) ──▶ Provider (~1,080 tok, up to 95% saved) ``` Default stacked combo runs `RTK → Caveman`. When both act on the same tool/context payload, savings compound: @@ -733,7 +751,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo **Will I be charged by OmniRoute?** No — it's free, open-source software on your machine. You only pay paid providers directly. OmniRoute has no billing system. **Are FREE providers really unlimited?** Yes — Kiro, Qoder, Pollinations, LongCat, Cloudflare. No catch. **Will compression hurt quality?** No — it only compresses the **input**; code, URLs, JSON are always protected. -**Does it work where AI is blocked?** Yes — 3-level proxy + 1proxy marketplace reach all 177 providers. +**Does it work where AI is blocked?** Yes — 3-level proxy + 1proxy marketplace reach all 226 providers. 📖 [User Guide](docs/guides/USER_GUIDE.md) · [API Reference](docs/reference/API_REFERENCE.md) · [Environment Config](docs/reference/ENVIRONMENT.md) @@ -803,7 +821,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo - **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) - **Streaming**: Server-Sent Events (SSE) + WebSocket bridge (`/v1/ws`) - **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization -- **Testing**: Node.js test runner + Vitest (**4,690+ test cases** across 517 files — unit, integration, E2E, security, ecosystem) +- **Testing**: Node.js test runner + Vitest (**14,965 test cases** across 517 files — unit, integration, E2E, security, ecosystem) - **Platforms**: Desktop (Electron), Android (Termux), PWA (any browser) - **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) - **Website**: [omniroute.online](https://omniroute.online) @@ -878,7 +896,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo | [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | | [i18n Guide](docs/guides/I18N.md) | 40+ language support, translation workflow, RTL | | [Release Checklist](docs/ops/RELEASE_CHECKLIST.md) | Pre-release validation steps | -| [Coverage Plan](docs/ops/COVERAGE_PLAN.md) | Test coverage strategy and 4,690+ test suite | +| [Coverage Plan](docs/ops/COVERAGE_PLAN.md) | Test coverage strategy and 14,965 test suite |
diff --git a/.license-allowlist.json b/config/quality/.license-allowlist.json similarity index 100% rename from .license-allowlist.json rename to config/quality/.license-allowlist.json diff --git a/complexity-baseline.json b/config/quality/complexity-baseline.json similarity index 100% rename from complexity-baseline.json rename to config/quality/complexity-baseline.json diff --git a/dependency-allowlist.json b/config/quality/dependency-allowlist.json similarity index 100% rename from dependency-allowlist.json rename to config/quality/dependency-allowlist.json diff --git a/duplication-baseline.json b/config/quality/duplication-baseline.json similarity index 100% rename from duplication-baseline.json rename to config/quality/duplication-baseline.json diff --git a/file-size-baseline.json b/config/quality/file-size-baseline.json similarity index 90% rename from file-size-baseline.json rename to config/quality/file-size-baseline.json index 7dbcc932a70..142b9b8f141 100644 --- a/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -2,9 +2,13 @@ "_comment": "Catraca de tamanho (check-file-size.mjs). frozen so pode encolher; arquivos novos <= cap. --update ratcheta.", "_rebaseline_v3.8.25": "Drift consciente do ciclo v3.8.24->v3.8.25 (features #3799-#3806: free-provider-rankings, plugins menu, proxy IP-family selector). 3 arquivos cresceram por feature legitima, nao por regressao de qualidade: ProxyRegistryManager.tsx 1072->1089, sidebarVisibility.ts 990->1006, schemas.ts 2519->2522. Encolher fica como debt para um refactor dedicado.", "_rebaseline_2026_06_15_3860_compression_ui": "PR #3860 own growth: sidebarVisibility.ts 1006->1100 (+94 = Compression Hub menu entries: Hub + per-engine Lite/Aggressive/Ultra pages + combos editor) and chatCore.ts 5812->5815 (+3 = compression UI config wiring). Cohesive feature growth, not a quality regression.", + "_rebaseline_2026_06_15_3885_glm_5_2": "PR #3885 own growth: pricing.ts 1508->1529 (+21 = GLM-5.2 pricing rows for glm-5.2 + effort aliases glm-5.2-high/-max, same $1.2/$5 schedule as glm-5.1; pure data). Also adds glm-5.2 specs to glmProvider.ts/modelSpecs.ts (modelSpecs.ts stays under cap). Cohesive model registration; not extractable.", + "_rebaseline_2026_06_15_3870_alias_lookup": "PR #3870 own growth: providerRegistry.ts 4703->4708 (+5 = generateModels() also stores each provider's models under its raw id, not only its alias, so getProviderModels(rawId) works when alias != id e.g. github->gh; preserves the existing first-wins guard). Cohesive registry fix; not extractable.", + "_rebaseline_2026_06_15_3846_sticky_combo_rr": "PR #3846 own growth: combo.ts 5204->5277 (+73 = combo-level sticky round-robin reusing the existing global stickyRoundRobinLimit knob #3847 added for account fallback: rrStickyTargets map + clampStickyRoundRobinTargetLimit + getStickyRoundRobinStartIndex/recordStickyRoundRobinSuccess helpers wired into handleRoundRobinCombo, with sticky-eviction tied to rrCounters eviction). Cohesive routing logic in the combo handler; not a movable block. Structural shrink of combo.ts tracked in #3501.", + "_rebaseline_2026_06_15_3871_empty_pool": "PR #3871 own growth: combo.ts 5203->5204 (+1 = guard expandAutoComboCandidatePool against an empty candidatePool array — Array.isArray(pool) && pool.length > 0 so [] falls through to active-connection expansion instead of early-returning). One-line correctness fix; not extractable.", "cap": 800, "frozen": { - "open-sse/config/providerRegistry.ts": 4703, + "open-sse/config/providerRegistry.ts": 4708, "open-sse/executors/antigravity.ts": 1649, "open-sse/executors/base.ts": 1218, "open-sse/executors/chatgpt-web.ts": 2870, @@ -30,14 +34,14 @@ "open-sse/services/batchProcessor.ts": 828, "open-sse/services/browserBackedChat.ts": 850, "open-sse/services/claudeCodeCompatible.ts": 1202, - "open-sse/services/combo.ts": 5203, + "open-sse/services/combo.ts": 5277, "open-sse/services/rateLimitManager.ts": 1017, "open-sse/services/tokenRefresh.ts": 1997, "open-sse/services/usage.ts": 3408, "open-sse/translator/request/openai-to-gemini.ts": 844, "open-sse/translator/response/openai-responses.ts": 878, "open-sse/utils/cursorAgentProtobuf.ts": 1521, - "open-sse/utils/stream.ts": 2710, + "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1385, "src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx": 1020, "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 2909, @@ -99,13 +103,15 @@ "src/shared/components/RequestLoggerV2.tsx": 1282, "src/shared/components/analytics/charts.tsx": 1558, "src/shared/constants/cliTools.ts": 875, - "src/shared/constants/pricing.ts": 1508, + "src/shared/constants/pricing.ts": 1529, "src/shared/constants/providers.ts": 3147, "src/shared/constants/sidebarVisibility.ts": 1100, "src/shared/services/cliRuntime.ts": 1090, "src/shared/validation/schemas.ts": 2523, "src/sse/handlers/chat.ts": 1425, - "src/sse/services/auth.ts": 2216 + "src/sse/services/auth.ts": 2216, + "open-sse/utils/stream/streamCore.ts": 2216, + "open-sse/utils/stream.ts": 2584 }, "_rebaseline_2026_06_09": "Re-baseline consciente pre-release v3.8.19: 9 arquivos cresceram durante o ciclo (features mergeadas: RequestLoggerV2 +281 request-logger rework, stream +101, combo +73, chatCore +45, catalog +32 fable-5/catalog-flag, callLogs +4, accountFallback +2, usageHistory novo 840) + core.ts +7 (fix resetAllDbModuleState, PR 3536). A catraca segue valendo destes valores — proximo crescimento falha. Decisao: encolher (esp. RequestLoggerV2/chatCore) e a issue #3501 ficam para o ciclo seguinte.", "_rebaseline_2026_06_11_phase1f": "Phase 1f (#3501): ProviderDetailPageClient.tsx 4948→4062 (-886 LOC); 3 novos hooks extraídos. useProviderConnections.ts=954 acima do cap=800 — justificado: extração direta do god-component (zero lógica nova), própria redução do cliente supera o custo. useProviderSettings.ts=263 e useProviderModels.ts=154 já abaixo do cap.", diff --git a/quality-baseline.json b/config/quality/quality-baseline.json similarity index 100% rename from quality-baseline.json rename to config/quality/quality-baseline.json diff --git a/test-discovery-baseline.json b/config/quality/test-discovery-baseline.json similarity index 100% rename from test-discovery-baseline.json rename to config/quality/test-discovery-baseline.json diff --git a/docs/architecture/REPOSITORY_MAP.md b/docs/architecture/REPOSITORY_MAP.md index bddfefd9d56..c5333921082 100644 --- a/docs/architecture/REPOSITORY_MAP.md +++ b/docs/architecture/REPOSITORY_MAP.md @@ -1,13 +1,13 @@ --- title: "Repository Map" -version: 3.8.2 -lastUpdated: 2026-05-13 +version: 3.8.26 +lastUpdated: 2026-06-15 --- # Repository Map > **One-line description for every directory and root file.** -> Last updated: 2026-05-13 — OmniRoute v3.8.0 +> Last updated: 2026-06-15 — OmniRoute v3.8.26 > > Use this map to navigate the codebase quickly. For deep dives, follow links to dedicated docs. @@ -23,8 +23,13 @@ OmniRoute/ ├── docs/ # Public documentation (you are here) ├── tests/ # All test suites (unit, integration, e2e, protocols-e2e) ├── public/ # Next.js static assets, PWA manifest, service worker, icons -├── config/ # Static config files +├── config/ # Static config + quality-gate state (i18n, payloadRules, quality/) ├── images/ # Marketing / README image assets +├── @omniroute/ # Publishable companion packages (opencode-plugin, opencode-provider) +├── skills/ # CLI/agent skill packs (cli-* + omni-* + config-codex-cli) +├── examples/ # Sample plugins + omniroute-cmd-hello starter +├── contrib/ # Community contributions (podman/) +├── .source/ # Fumadocs source config (source.config.mjs + server/browser/dynamic) ├── .github/ # GitHub Actions workflows + issue templates + PR template ├── .husky/ # Git hooks (pre-commit, pre-push) ├── .claude/ # Claude Code slash commands (project-scoped) @@ -34,6 +39,7 @@ OmniRoute/ ├── _mono_repo/ # Historic subprojects (cloud, site, vscode-extension) ├── _references/ # Read-only reference clones from related OSS projects ├── _tasks/ # Per-release task tracking files (informal) +├── .build/ .worktrees/ dist/ # local build / git-worktree / build-output scratch (gitignored) ├── .issues/ # Local issue cache (gitignored) ├── .playwright-mcp/ # Playwright MCP test artifacts ├── coverage/ # c8 coverage output (gitignored) @@ -48,45 +54,63 @@ OmniRoute/ ## Root files -| File | Purpose | -| ------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- | -| **README.md** | Marketing landing page + quick start + feature matrix (see also `llm.txt`) | -| **CHANGELOG.md** | Per-release changelog (auto-generated by `/version-bump-cc` skill) | -| **LICENSE** | MIT license text | -| **CLAUDE.md** | Project rules for Claude Code agents (hard rules, conventions, scenarios) | -| **AGENTS.md** | Same as CLAUDE.md but for non-Claude AI agents (Codex, Cursor, etc.) | -| **GEMINI.md** | Concise rules for Gemini-based agents (subset of CLAUDE.md) | -| **CONTRIBUTING.md** | Contributor guide: setup, conventional commits, testing, PR flow | -| **SECURITY.md** | Vulnerability reporting policy, supported versions, threat model | -| **CODE_OF_CONDUCT.md** | Contributor Covenant — community behavior expectations | -| **llm.txt** | Plain-text landing optimized for LLM crawlers (SEO for AI assistants) | -| **Tuto_Qdrant.md** | Tutorial for enabling Qdrant vector memory — **integration currently dormant** (see banner; primary memory docs in `docs/frameworks/MEMORY.md`) | -| **package.json** | npm manifest, scripts, dependencies, engines, c8 coverage gate | -| **package-lock.json** | Locked dependency tree | -| **tsconfig.json** | Root TypeScript config | -| **tsconfig.typecheck-core.json** | Typecheck config for `src/` core | -| **tsconfig.typecheck-noimplicit-core.json** | Strict (`noImplicitAny`) typecheck | -| **tsconfig.tsbuildinfo** | TS incremental build cache (gitignored) | -| **next.config.mjs** | Next.js 16 build configuration (standalone output) | -| **next-env.d.ts** | Next.js auto-generated env types | -| **eslint.config.mjs** | ESLint flat config (rules per project area) | -| **prettier.config.mjs** | Prettier formatting rules | -| **postcss.config.mjs** | PostCSS config for Tailwind/CSS pipeline | -| **playwright.config.ts** | Playwright E2E test config | -| **vitest.config.ts** | Vitest config (default suite) | -| **vitest.mcp.config.ts** | Vitest config for MCP server / autoCombo / cache suites | -| **sonar-project.properties** | SonarQube/SonarCloud config (code quality) | -| **Dockerfile** | Multi-stage Docker build (builder → runner-base → runner-cli) | -| **docker-compose.yml** | Dev compose with 4 profiles (base, cli, host, cliproxyapi) + redis sidecar | -| **docker-compose.prod.yml** | Production compose (port 20130, redis, named volumes) | -| **.dockerignore** | Files excluded from Docker context | -| **fly.toml** | Fly.io deployment config (region `sin`, port 20128, /data volume) | -| **.env.example** | Template env file (815 lines, auto-copied to `.env` on first install) | -| **.gitignore** | Git ignore patterns | -| **.npmignore** | npm publish exclusion list | -| **.npmrc** | npm config (registry, lockfile policy) | -| **.node-version** | Node version pin (used by nvm-compatible tools) | -| **.nvmrc** | Node version pin for nvm | +| File | Purpose | +| ------------------------------------------- | ---------------------------------------------------------------------------------------- | +| **README.md** | Marketing landing page + quick start + feature matrix (see also `llm.txt`) | +| **CHANGELOG.md** | Per-release changelog (auto-generated by `/version-bump-cc` skill) | +| **LICENSE** | MIT license text | +| **CLAUDE.md** | Project rules for Claude Code agents (hard rules, conventions, scenarios) | +| **AGENTS.md** | Same as CLAUDE.md but for non-Claude AI agents (Codex, Cursor, etc.) | +| **GEMINI.md** | Concise rules for Gemini-based agents (subset of CLAUDE.md) | +| **CONTRIBUTING.md** | Contributor guide: setup, conventional commits, testing, PR flow | +| **SECURITY.md** | Vulnerability reporting policy, supported versions, threat model | +| **CODE_OF_CONDUCT.md** | Contributor Covenant — community behavior expectations | +| **llm.txt** | Plain-text landing optimized for LLM crawlers (SEO for AI assistants) | +| **package.json** | npm manifest, scripts, dependencies, engines, c8 coverage gate | +| **package-lock.json** | Locked dependency tree | +| **tsconfig.json** | Root TypeScript config | +| **tsconfig.typecheck-core.json** | Typecheck config for `src/` core | +| **tsconfig.typecheck-noimplicit-core.json** | Strict (`noImplicitAny`) typecheck | +| **tsconfig.tsbuildinfo** | TS incremental build cache (gitignored) | +| **next.config.mjs** | Next.js 16 build configuration (standalone output) | +| **next-env.d.ts** | Next.js auto-generated env types | +| **eslint.config.mjs** | ESLint flat config (rules per project area) | +| **prettier.config.mjs** | Prettier formatting rules | +| **postcss.config.mjs** | PostCSS config for Tailwind/CSS pipeline | +| **playwright.config.ts** | Playwright E2E test config | +| **vitest.config.ts** | Vitest config (default suite) | +| **vitest.mcp.config.ts** | Vitest config for MCP server / autoCombo / cache suites | +| **sonar-project.properties** | SonarQube/SonarCloud config (code quality) | +| **Dockerfile** | Multi-stage Docker build (builder → runner-base → runner-cli) | +| **docker-compose.yml** | Dev compose with 4 profiles (base, cli, host, cliproxyapi) + redis sidecar | +| **docker-compose.prod.yml** | Production compose (port 20130, redis, named volumes) | +| **.dockerignore** | Files excluded from Docker context | +| **fly.toml** | Fly.io deployment config (region `sin`, port 20128, /data volume) | +| **.env.example** | Template env file (auto-copied to `.env` on first install) | +| **.gitignore** | Git ignore patterns | +| **.npmignore** | npm publish exclusion list | +| **.npmrc** | npm config (registry, lockfile policy) | +| **.node-version** | Node version pin (used by nvm-compatible tools) | +| **.nvmrc** | Node version pin for nvm | +| **eslint.complexity.config.mjs** | ESLint config for the complexity ratchet (`scripts/check/check-complexity.mjs --config`) | +| **eslint.sonarjs.config.mjs** | ESLint config for SonarJS rules (cognitive complexity / duplication) | +| **source.config.ts** | Fumadocs `defineDocs` source config (feeds `.source/`) | +| **knip.json** | Knip config — unused files/exports/deps (feeds the dead-code gate) | +| **stryker.conf.json** | Stryker mutation-testing config | +| **.size-limit.json** | size-limit bundle budget config | +| **semcheck.yaml** | semcheck (spec↔code drift) config | +| **promptfooconfig.yaml** | promptfoo eval config | +| **.gitleaks.toml** | gitleaks secret-scan ruleset | +| **.zizmor.yml** | zizmor GitHub-Actions security-lint config | +| **socket.yml** | Socket.dev supply-chain config | +| **news.json** | In-app release-notes feed (read by `src/shared/utils/releaseNotes.ts`) | +| **flake.nix** / **flake.lock** | Nix dev-shell definition + lock | +| **.env** | Local secrets (gitignored — generated from `.env.example`) | + +> **Moved out of the root in v3.8.26 (declutter):** +> +> - **→ `config/quality/`:** `quality-baseline.json`, `complexity-baseline.json`, `duplication-baseline.json`, `file-size-baseline.json`, `test-discovery-baseline.json`, `dependency-allowlist.json`, `.license-allowlist.json`, and the generated `quality-metrics.json` (gitignored). See [`## config/`](#config--static-configs--quality-gate-state). +> - **→ `docs/ops/`:** `DOCUMENTATION_AUDIT_REPORT.md`. --- @@ -117,84 +141,84 @@ src/ ### `src/app/` — App Router (Next.js 16) -| Path | Purpose | -| ---------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `app/api/v1/` | Public OpenAI-compat API (~25 sub-routes: chat, completions, embeddings, files, batches, audio, images, videos, music, rerank, moderations, search, ws, agents, accounts, providers, etc.) | -| `app/api/v1beta/` | Gemini-style API endpoints | -| `app/api/playground/` | Playground Studio routes: `improve-prompt/` (POST — LLM prompt rewriter), `presets/` (GET list / POST create), `presets/[id]/` (GET / PUT / DELETE) — see `docs/frameworks/PLAYGROUND_STUDIO.md` | -| `app/api/` (non-v1) | Management/admin routes (~60 directories: providers, combos, settings, mcp, a2a, evals, memory, skills, webhooks, compliance, resilience, monitoring, tunnels, cli-tools, etc.) | -| `app/api/tools/agent-bridge/` | AgentBridge REST API — 12 routes (server control, agent state/DNS/mappings, bypass, cert, upstream-CA). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/AGENTBRIDGE.md §7`. | -| `app/api/tools/traffic-inspector/` | Traffic Inspector REST + WS API — 16+ routes (requests, sessions, hosts, capture-modes, export, ws). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/TRAFFIC_INSPECTOR.md §8`. | -| `app/a2a/` | A2A JSON-RPC 2.0 entry point (`POST /a2a`) | -| `app/.well-known/agent.json/` | A2A Agent Card (discovery) | -| `app/(dashboard)/dashboard/` | Dashboard UI pages (~35 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, activity, etc.) | -| `app/(dashboard)/dashboard/search-tools/` | Search Tools Studio UI (3 tabs: Search/Scrape/Compare + SearchConceptCard + ProviderCatalog) — see `docs/frameworks/SEARCH_TOOLS_STUDIO.md` | -| `app/(dashboard)/dashboard/` | Dashboard UI pages (~30 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, etc.) | +| Path | Purpose | +| ---------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `app/api/v1/` | Public OpenAI-compat API (~25 sub-routes: chat, completions, embeddings, files, batches, audio, images, videos, music, rerank, moderations, search, ws, agents, accounts, providers, etc.) | +| `app/api/v1beta/` | Gemini-style API endpoints | +| `app/api/playground/` | Playground Studio routes: `improve-prompt/` (POST — LLM prompt rewriter), `presets/` (GET list / POST create), `presets/[id]/` (GET / PUT / DELETE) — see `docs/frameworks/PLAYGROUND_STUDIO.md` | +| `app/api/` (non-v1) | Management/admin routes (~60 directories: providers, combos, settings, mcp, a2a, evals, memory, skills, webhooks, compliance, resilience, monitoring, tunnels, cli-tools, etc.) | +| `app/api/tools/agent-bridge/` | AgentBridge REST API — 12 routes (server control, agent state/DNS/mappings, bypass, cert, upstream-CA). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/AGENTBRIDGE.md §7`. | +| `app/api/tools/traffic-inspector/` | Traffic Inspector REST + WS API — 16+ routes (requests, sessions, hosts, capture-modes, export, ws). LOCAL_ONLY + SPAWN_CAPABLE. See `docs/frameworks/TRAFFIC_INSPECTOR.md §8`. | +| `app/a2a/` | A2A JSON-RPC 2.0 entry point (`POST /a2a`) | +| `app/.well-known/agent.json/` | A2A Agent Card (discovery) | +| `app/(dashboard)/dashboard/` | Dashboard UI pages (~35 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, activity, etc.) | +| `app/(dashboard)/dashboard/search-tools/` | Search Tools Studio UI (3 tabs: Search/Scrape/Compare + SearchConceptCard + ProviderCatalog) — see `docs/frameworks/SEARCH_TOOLS_STUDIO.md` | +| `app/(dashboard)/dashboard/` | Dashboard UI pages (~30 pages: providers, combos, settings, memory, skills, webhooks, evals, audit, batch, cache, costs, health, system, etc.) | | `app/(dashboard)/dashboard/memory/` | Memory Studio (plan 21): `page.tsx` (3-tab shell), `components/` (MemoryConceptCard, MemoryEngineStatus, EmbeddingSourceSelector, EditMemoryModal, RetrievePreview, QdrantConfigCard, RerankConfigCard), `components/tabs/` (MemoriesTab, PlaygroundTab, EngineTab), `hooks/` (useEngineStatus, useMemorySettings) | -| `app/(dashboard)/dashboard/tools/agent-bridge/` | AgentBridge dashboard page — server card, 9 agent cards, setup wizard, model mapping, bypass list. i18n PT-BR + EN. See `docs/frameworks/AGENTBRIDGE.md`. | -| `app/(dashboard)/dashboard/tools/traffic-inspector/` | Traffic Inspector dashboard page — DevTools split, 7 detail tabs, 4 capture mode toggles, session recorder, context colorization. i18n PT-BR + EN. See `docs/frameworks/TRAFFIC_INSPECTOR.md`. | -| `app/(dashboard)/dashboard/activity/` | Activity feed page (Group B): `page.tsx` (server) + `ActivityFeedClient.tsx` + `components/{ActivityFeed,ActivityItem,DayHeader,EventTypeFilter}.tsx` — see `docs/architecture/MONITORING_SECTIONS.md` | -| `app/(dashboard)/dashboard/costs/quota-share/` | Quota Sharing page (Group B): `QuotaSharePageClient.tsx` + `components/{PoolCard,DimensionBar,AllocationTable,BurnRateChart,QuotaConceptCard,CreatePoolModal,EditAllocationsModal}.tsx` + `hooks/{usePools,usePoolUsage,useLocalStoragePoolMigration}.ts` | -| `app/(dashboard)/dashboard/costs/quota-share/plans/` | Provider plan config page (Group B): `page.tsx` + `ProviderPlanConfigClient.tsx` — quota dimensions per connection override | -| `app/docs/` | Embedded documentation viewer (renders `docs/*.md`) | -| `app/landing/` | Marketing landing page | -| `app/login/`, `forgot-password/`, `forbidden/` | Auth-related pages | -| `app/{400,401,403,408,429,500,502,503}/` | HTTP error pages | -| `app/maintenance/`, `offline/`, `status/`, `privacy/`, `terms/`, `callback/` | Static/status pages | -| `app/layout.tsx`, `page.tsx`, `manifest.ts`, `globals.css` | Root layout, home, PWA manifest, global CSS | -| `app/error.tsx`, `global-error.tsx`, `not-found.tsx`, `loading.tsx` | Error boundaries | +| `app/(dashboard)/dashboard/tools/agent-bridge/` | AgentBridge dashboard page — server card, 9 agent cards, setup wizard, model mapping, bypass list. i18n PT-BR + EN. See `docs/frameworks/AGENTBRIDGE.md`. | +| `app/(dashboard)/dashboard/tools/traffic-inspector/` | Traffic Inspector dashboard page — DevTools split, 7 detail tabs, 4 capture mode toggles, session recorder, context colorization. i18n PT-BR + EN. See `docs/frameworks/TRAFFIC_INSPECTOR.md`. | +| `app/(dashboard)/dashboard/activity/` | Activity feed page (Group B): `page.tsx` (server) + `ActivityFeedClient.tsx` + `components/{ActivityFeed,ActivityItem,DayHeader,EventTypeFilter}.tsx` — see `docs/architecture/MONITORING_SECTIONS.md` | +| `app/(dashboard)/dashboard/costs/quota-share/` | Quota Sharing page (Group B): `QuotaSharePageClient.tsx` + `components/{PoolCard,DimensionBar,AllocationTable,BurnRateChart,QuotaConceptCard,CreatePoolModal,EditAllocationsModal}.tsx` + `hooks/{usePools,usePoolUsage,useLocalStoragePoolMigration}.ts` | +| `app/(dashboard)/dashboard/costs/quota-share/plans/` | Provider plan config page (Group B): `page.tsx` + `ProviderPlanConfigClient.tsx` — quota dimensions per connection override | +| `app/docs/` | Embedded documentation viewer (renders `docs/*.md`) | +| `app/landing/` | Marketing landing page | +| `app/login/`, `forgot-password/`, `forbidden/` | Auth-related pages | +| `app/{400,401,403,408,429,500,502,503}/` | HTTP error pages | +| `app/maintenance/`, `offline/`, `status/`, `privacy/`, `terms/`, `callback/` | Static/status pages | +| `app/layout.tsx`, `page.tsx`, `manifest.ts`, `globals.css` | Root layout, home, PWA manifest, global CSS | +| `app/error.tsx`, `global-error.tsx`, `not-found.tsx`, `loading.tsx` | Error boundaries | ### `src/lib/` — Core libraries (~50 modules) -| Module | Purpose | -| ---------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `a2a/` | A2A protocol task manager, skills (5), streaming | -| `acp/` | CLI Agent Registry (local CLI discovery — see `docs/frameworks/AGENT_PROTOCOLS_GUIDE.md`) | -| `api/` | Shared API helpers (`requireManagementAuth`, validation) | -| `auth/` | Session, password hashing, token validation | -| `batches/` | OpenAI Batches API handlers | -| `catalog/` | Provider catalog Zod validation + capability resolution | -| `cloudAgent/` | Cloud Agents (Codex Cloud, Devin, Jules) — see `docs/frameworks/CLOUD_AGENT.md` | -| `combos/` | Combo resolution + reorder helpers | -| `audit/` | Activity feed helpers: `highLevelActions.ts` (allowlist + `isHighLevelAction()`), `activityIcons.ts` (action → icon/verb map), `timeline.ts` (groupByDay/relativeTime) — see `docs/architecture/MONITORING_SECTIONS.md` | -| `compliance/` | Audit log + provider audit — see `docs/security/COMPLIANCE.md` | -| `compression/` | Compression engine glue (engines live in `open-sse/services/compression/`) | -| `config/` | Runtime config helpers | -| `db/` | 45+ domain DB modules + 55 migrations (always go through here for SQLite) | +| Module | Purpose | +| ---------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `a2a/` | A2A protocol task manager, skills (5), streaming | +| `acp/` | CLI Agent Registry (local CLI discovery — see `docs/frameworks/AGENT_PROTOCOLS_GUIDE.md`) | +| `api/` | Shared API helpers (`requireManagementAuth`, validation) | +| `auth/` | Session, password hashing, token validation | +| `batches/` | OpenAI Batches API handlers | +| `catalog/` | Provider catalog Zod validation + capability resolution | +| `cloudAgent/` | Cloud Agents (Codex Cloud, Devin, Jules) — see `docs/frameworks/CLOUD_AGENT.md` | +| `combos/` | Combo resolution + reorder helpers | +| `audit/` | Activity feed helpers: `highLevelActions.ts` (allowlist + `isHighLevelAction()`), `activityIcons.ts` (action → icon/verb map), `timeline.ts` (groupByDay/relativeTime) — see `docs/architecture/MONITORING_SECTIONS.md` | +| `compliance/` | Audit log + provider audit — see `docs/security/COMPLIANCE.md` | +| `compression/` | Compression engine glue (engines live in `open-sse/services/compression/`) | +| `config/` | Runtime config helpers | +| `db/` | 45+ domain DB modules + 55 migrations (always go through here for SQLite) | | `quota/` | Quota Sharing Engine: `dimensions.ts` (types/Zod), `types.ts` (QuotaStore interface), `sqliteQuotaStore.ts`, `redisQuotaStore.ts`, `storeFactory.ts`, `fairShare.ts`, `burnRate.ts`, `planResolver.ts`, `planRegistry.ts`, `saturationSignals.ts`, `enforce.ts`, `spendRecorder.ts` — see `docs/routing/QUOTA_SHARE.md` | -| `display/` | UI formatting helpers (cost, latency, etc.) | -| `embeddings/` | Embeddings service helpers | -| `env/` | Env variable parsing + validation | -| `evals/` | Eval framework (suites, runner, runtime) — see `docs/frameworks/EVALS.md` | -| `guardrails/` | PII masker, prompt injection, vision bridge — see `docs/security/GUARDRAILS.md` | -| `jobs/` | Background jobs (cron-like) | -| `memory/` | Conversational memory (SQLite FTS5 + sqlite-vec hybrid RRF + Qdrant tier 2) — see `docs/frameworks/MEMORY.md` | -| `memory/embedding/` | Multi-source embedding layer: `index.ts` (resolver), `remote.ts`, `staticPotion.ts`, `transformersLocal.ts`, `cache.ts`, `types.ts` (plan 21) | -| `memory/vectorStore.ts` | sqlite-vec v0.1.9 wrapper — KNN brute-force + hybrid RRF (FTS5 + vector, k=60). Lazy-init, degrades gracefully when sqlite-vec unavailable. (plan 21) | -| `memory/reindex.ts` | `runReindexBatch()` — processes memories with `needs_reindex=1` in background; called by `POST /api/memory/reindex` and lazy-backfill path. (plan 21) | -| `monitoring/` | Health checks, metrics emission | -| `oauth/` | OAuth flows for 14 providers (claude, codex, antigravity, cursor, github, gemini, kimi-coding, kilocode, cline, qwen, kiro, qoder, gitlab-duo, windsurf) | -| `plugins/` | Plugin registry | -| `promptCache/` | Anthropic-style prompt cache breakpoints | -| `skills/` | Skills framework (built-in + marketplace + SkillsSH) — see `docs/frameworks/SKILLS.md` | -| `playground/` | Playground Studio shared helpers: `codeExport.ts` (curl/Python/TS generator), `promptImprover.ts` (meta-prompt builder), `streamMetrics.ts` (pure TTFT/TPS), `types.ts` (pricing table) — see `docs/frameworks/PLAYGROUND_STUDIO.md` | -| `webhookDispatcher.ts` | HMAC webhook delivery — see `docs/frameworks/WEBHOOKS.md` | -| `cloudflaredTunnel.ts`, `ngrokTunnel.ts` | Tunnel managers — see `docs/ops/TUNNELS_GUIDE.md` | -| `oneproxySync.ts`, `oneproxyRotator.ts` | 1proxy free proxy marketplace — see `docs/ops/PROXY_GUIDE.md` | -| `cloudSync.ts`, `initCloudSync.ts` | Optional cloud sync of state | -| `localDb.ts` | Re-export barrel for db modules (no logic — re-exports only) | -| `cacheLayer.ts`, `idempotencyLayer.ts` | Request caching + idempotency | -| (~30 more top-level files) | Specialized helpers (logEnv, modelsDevSync, piiSanitizer, etc.) | +| `display/` | UI formatting helpers (cost, latency, etc.) | +| `embeddings/` | Embeddings service helpers | +| `env/` | Env variable parsing + validation | +| `evals/` | Eval framework (suites, runner, runtime) — see `docs/frameworks/EVALS.md` | +| `guardrails/` | PII masker, prompt injection, vision bridge — see `docs/security/GUARDRAILS.md` | +| `jobs/` | Background jobs (cron-like) | +| `memory/` | Conversational memory (SQLite FTS5 + sqlite-vec hybrid RRF + Qdrant tier 2) — see `docs/frameworks/MEMORY.md` | +| `memory/embedding/` | Multi-source embedding layer: `index.ts` (resolver), `remote.ts`, `staticPotion.ts`, `transformersLocal.ts`, `cache.ts`, `types.ts` (plan 21) | +| `memory/vectorStore.ts` | sqlite-vec v0.1.9 wrapper — KNN brute-force + hybrid RRF (FTS5 + vector, k=60). Lazy-init, degrades gracefully when sqlite-vec unavailable. (plan 21) | +| `memory/reindex.ts` | `runReindexBatch()` — processes memories with `needs_reindex=1` in background; called by `POST /api/memory/reindex` and lazy-backfill path. (plan 21) | +| `monitoring/` | Health checks, metrics emission | +| `oauth/` | OAuth flows for 14 providers (claude, codex, antigravity, cursor, github, gemini, kimi-coding, kilocode, cline, qwen, kiro, qoder, gitlab-duo, windsurf) | +| `plugins/` | Plugin registry | +| `promptCache/` | Anthropic-style prompt cache breakpoints | +| `skills/` | Skills framework (built-in + marketplace + SkillsSH) — see `docs/frameworks/SKILLS.md` | +| `playground/` | Playground Studio shared helpers: `codeExport.ts` (curl/Python/TS generator), `promptImprover.ts` (meta-prompt builder), `streamMetrics.ts` (pure TTFT/TPS), `types.ts` (pricing table) — see `docs/frameworks/PLAYGROUND_STUDIO.md` | +| `webhookDispatcher.ts` | HMAC webhook delivery — see `docs/frameworks/WEBHOOKS.md` | +| `cloudflaredTunnel.ts`, `ngrokTunnel.ts` | Tunnel managers — see `docs/ops/TUNNELS_GUIDE.md` | +| `oneproxySync.ts`, `oneproxyRotator.ts` | 1proxy free proxy marketplace — see `docs/ops/PROXY_GUIDE.md` | +| `cloudSync.ts`, `initCloudSync.ts` | Optional cloud sync of state | +| `localDb.ts` | Re-export barrel for db modules (no logic — re-exports only) | +| `cacheLayer.ts`, `idempotencyLayer.ts` | Request caching + idempotency | +| (~30 more top-level files) | Specialized helpers (logEnv, modelsDevSync, piiSanitizer, etc.) | ### `src/db/` — Database (45+ modules + 55 migrations) -| Subdir | Purpose | -| ---------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `db/core.ts` | `getDbInstance()` singleton with WAL journaling | -| `db/migrations/` | Versioned SQL files (idempotent, transactional). `073_memory_vec.sql` adds `memory_vec_meta` + `needs_reindex` column (plan 21). | -| `db/playgroundPresets.ts` | CRUD module for Playground Studio presets (`listPlaygroundPresets`, `getPlaygroundPreset`, `createPlaygroundPreset`, `updatePlaygroundPreset`, `deletePlaygroundPreset`) | -| `db/memoryVec.ts`| CRUD for `memory_vec_meta` (active_dim, embedding_signature, last_reset_at, vec_loaded) + `markMemoryNeedsReindex`, `getMemoryReindexQueue`, etc. (plan 21) | -| `db/.ts` | One module per domain: providers, combos, apiKeys, users, sessions, usage, audit*log, webhooks, skills, memory_entries, cloud_agent_tasks, evals*\*, reasoning_cache, etc. | +| Subdir | Purpose | +| ------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `db/core.ts` | `getDbInstance()` singleton with WAL journaling | +| `db/migrations/` | Versioned SQL files (idempotent, transactional). `073_memory_vec.sql` adds `memory_vec_meta` + `needs_reindex` column (plan 21). | +| `db/playgroundPresets.ts` | CRUD module for Playground Studio presets (`listPlaygroundPresets`, `getPlaygroundPreset`, `createPlaygroundPreset`, `updatePlaygroundPreset`, `deletePlaygroundPreset`) | +| `db/memoryVec.ts` | CRUD for `memory_vec_meta` (active_dim, embedding_signature, last_reset_at, vec_loaded) + `markMemoryNeedsReindex`, `getMemoryReindexQueue`, etc. (plan 21) | +| `db/.ts` | One module per domain: providers, combos, apiKeys, users, sessions, usage, audit*log, webhooks, skills, memory_entries, cloud_agent_tasks, evals*\*, reasoning_cache, etc. | ### `src/domain/` @@ -455,9 +479,24 @@ open-sse/ --- -## `config/` — Static Configs - -Shipped configuration templates and sample files (referenced by setup wizard). +## `config/` — Static Configs + Quality-Gate State + +Shipped configuration templates plus the committed quality-gate baselines +(moved here from the repo root in v3.8.26 to keep the root lean). + +| Path | Purpose | +| --------------------------------------------- | -------------------------------------------------------------------------------- | +| `config/i18n.json` | Locale list + metadata (canonical source for the 42-locale count) | +| `config/i18n-schema.json` | JSON schema validating `i18n.json` | +| `config/payloadRules.json` | Upstream payload sanitization rules | +| `config/quality/quality-baseline.json` | Multi-metric ratchet baseline (`scripts/quality/check-quality-ratchet.mjs`) | +| `config/quality/complexity-baseline.json` | Frozen ESLint-complexity baseline (`check-complexity.mjs`) | +| `config/quality/duplication-baseline.json` | Frozen jscpd duplication baseline (`check-duplication.mjs`) | +| `config/quality/file-size-baseline.json` | Frozen per-file size baseline (`check-file-size.mjs`) | +| `config/quality/test-discovery-baseline.json` | Frozen orphan-test baseline (`check-test-discovery.mjs`) | +| `config/quality/dependency-allowlist.json` | Approved dependencies allowlist (`check-deps.mjs`) | +| `config/quality/.license-allowlist.json` | SPDX license allowlist (`check-licenses.mjs`) | +| `config/quality/quality-metrics.json` | Ephemeral collected metrics (generated by `collect-metrics.mjs`; **gitignored**) | --- @@ -484,15 +523,15 @@ Shipped configuration templates and sample files (referenced by setup wizard). ## `.claude/` — Claude Code Slash Commands -| File | Purpose | -| ----------------------------------------------------------------- | -------------------------------------------------- | -| `commands/version-bump-cc.md` | `/version-bump-cc` — bump version + auto-changelog | -| `commands/generate-release-cc.md` | `/generate-release-cc` — full release workflow | -| `commands/deploy-vps-{local,akamai,both}-cc.md` | Deploy to VPS | -| `commands/capture-release-evidences-cc.md` | Browser-record new features as WebP | -| `commands/review-{prs,discussions}-cc.md` | Triage GitHub PRs/discussions | -| `commands/{review-issues,implement-features}-cc.md` | Issue workflows | -| `settings.local.json` | Per-project Claude Code settings | +| File | Purpose | +| --------------------------------------------------- | -------------------------------------------------- | +| `commands/version-bump-cc.md` | `/version-bump-cc` — bump version + auto-changelog | +| `commands/generate-release-cc.md` | `/generate-release-cc` — full release workflow | +| `commands/deploy-vps-{local,akamai,both}-cc.md` | Deploy to VPS | +| `commands/capture-release-evidences-cc.md` | Browser-record new features as WebP | +| `commands/review-{prs,discussions}-cc.md` | Triage GitHub PRs/discussions | +| `commands/{review-issues,implement-features}-cc.md` | Issue workflows | +| `settings.local.json` | Per-project Claude Code settings | --- diff --git a/docs/guides/FEATURES.md b/docs/guides/FEATURES.md index df79c1d0ef5..0ae629121b2 100644 --- a/docs/guides/FEATURES.md +++ b/docs/guides/FEATURES.md @@ -53,6 +53,8 @@ The v3.7.x → v3.8.0 cycle added zero-config auto routing, new providers, OAuth Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. +OpenRouter connections can store a per-connection `preset` in Advanced Settings. When set, OmniRoute sends it as the OpenRouter top-level request field, for example `"preset": "email-copywriter"`, unless the client request already supplied its own `preset`. + ![Providers Dashboard](../screenshots/01-providers.png) --- diff --git a/docs/i18n/ar/CHANGELOG.md b/docs/i18n/ar/CHANGELOG.md index d4c2175cbfe..d2c25a654a8 100644 --- a/docs/i18n/ar/CHANGELOG.md +++ b/docs/i18n/ar/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/az/CHANGELOG.md b/docs/i18n/az/CHANGELOG.md index 4fb4d1ee043..ac25be0c6dc 100644 --- a/docs/i18n/az/CHANGELOG.md +++ b/docs/i18n/az/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/bg/CHANGELOG.md b/docs/i18n/bg/CHANGELOG.md index 4fb4d1ee043..ac25be0c6dc 100644 --- a/docs/i18n/bg/CHANGELOG.md +++ b/docs/i18n/bg/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/bn/CHANGELOG.md b/docs/i18n/bn/CHANGELOG.md index 47b5e3cec42..0834d753702 100644 --- a/docs/i18n/bn/CHANGELOG.md +++ b/docs/i18n/bn/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/cs/CHANGELOG.md b/docs/i18n/cs/CHANGELOG.md index ee25f9424af..4e03f9602b0 100644 --- a/docs/i18n/cs/CHANGELOG.md +++ b/docs/i18n/cs/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/da/CHANGELOG.md b/docs/i18n/da/CHANGELOG.md index 8f5040c48c1..fc5df4cbbaa 100644 --- a/docs/i18n/da/CHANGELOG.md +++ b/docs/i18n/da/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/de/CHANGELOG.md b/docs/i18n/de/CHANGELOG.md index b680e3dc980..1203fafe426 100644 --- a/docs/i18n/de/CHANGELOG.md +++ b/docs/i18n/de/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/es/CHANGELOG.md b/docs/i18n/es/CHANGELOG.md index 85dd13446cc..1fe9641e4fd 100644 --- a/docs/i18n/es/CHANGELOG.md +++ b/docs/i18n/es/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/fa/CHANGELOG.md b/docs/i18n/fa/CHANGELOG.md index 3060c2c78cf..b79d3e6d715 100644 --- a/docs/i18n/fa/CHANGELOG.md +++ b/docs/i18n/fa/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/fi/CHANGELOG.md b/docs/i18n/fi/CHANGELOG.md index d48ee8a418d..a012ba3fdbd 100644 --- a/docs/i18n/fi/CHANGELOG.md +++ b/docs/i18n/fi/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/fr/CHANGELOG.md b/docs/i18n/fr/CHANGELOG.md index 33e1682eb77..4ebd4e3b117 100644 --- a/docs/i18n/fr/CHANGELOG.md +++ b/docs/i18n/fr/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/gu/CHANGELOG.md b/docs/i18n/gu/CHANGELOG.md index 34c5b92c535..26119eb3b95 100644 --- a/docs/i18n/gu/CHANGELOG.md +++ b/docs/i18n/gu/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/he/CHANGELOG.md b/docs/i18n/he/CHANGELOG.md index dfb651c7e3b..cec30fe21a4 100644 --- a/docs/i18n/he/CHANGELOG.md +++ b/docs/i18n/he/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/hi/CHANGELOG.md b/docs/i18n/hi/CHANGELOG.md index 88193ed52d9..63b343e7412 100644 --- a/docs/i18n/hi/CHANGELOG.md +++ b/docs/i18n/hi/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/hu/CHANGELOG.md b/docs/i18n/hu/CHANGELOG.md index d8cba0e4c4f..703cc401fe1 100644 --- a/docs/i18n/hu/CHANGELOG.md +++ b/docs/i18n/hu/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/id/CHANGELOG.md b/docs/i18n/id/CHANGELOG.md index e89b56ac968..ccd63ed6d0c 100644 --- a/docs/i18n/id/CHANGELOG.md +++ b/docs/i18n/id/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/in/CHANGELOG.md b/docs/i18n/in/CHANGELOG.md index d8d45ea3543..fa54bd1fca5 100644 --- a/docs/i18n/in/CHANGELOG.md +++ b/docs/i18n/in/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/it/CHANGELOG.md b/docs/i18n/it/CHANGELOG.md index d592c0054a9..51c2d0fc795 100644 --- a/docs/i18n/it/CHANGELOG.md +++ b/docs/i18n/it/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ja/CHANGELOG.md b/docs/i18n/ja/CHANGELOG.md index 1a368244902..7c8b2f63175 100644 --- a/docs/i18n/ja/CHANGELOG.md +++ b/docs/i18n/ja/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ko/CHANGELOG.md b/docs/i18n/ko/CHANGELOG.md index 05d5e753db4..295d7b8e55e 100644 --- a/docs/i18n/ko/CHANGELOG.md +++ b/docs/i18n/ko/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/mr/CHANGELOG.md b/docs/i18n/mr/CHANGELOG.md index 19564c8d839..8f00e93e736 100644 --- a/docs/i18n/mr/CHANGELOG.md +++ b/docs/i18n/mr/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ms/CHANGELOG.md b/docs/i18n/ms/CHANGELOG.md index b4c645449f8..8a1c635018c 100644 --- a/docs/i18n/ms/CHANGELOG.md +++ b/docs/i18n/ms/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/nl/CHANGELOG.md b/docs/i18n/nl/CHANGELOG.md index 430ffc4ce6e..7cfe6406d96 100644 --- a/docs/i18n/nl/CHANGELOG.md +++ b/docs/i18n/nl/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/no/CHANGELOG.md b/docs/i18n/no/CHANGELOG.md index b14d9e9b9dd..4de0e1bf8da 100644 --- a/docs/i18n/no/CHANGELOG.md +++ b/docs/i18n/no/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/phi/CHANGELOG.md b/docs/i18n/phi/CHANGELOG.md index 7a69e46cd1e..2d0fd80b042 100644 --- a/docs/i18n/phi/CHANGELOG.md +++ b/docs/i18n/phi/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/pl/CHANGELOG.md b/docs/i18n/pl/CHANGELOG.md index 5a2b6c0af13..a6c0c4638c7 100644 --- a/docs/i18n/pl/CHANGELOG.md +++ b/docs/i18n/pl/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/pt-BR/CHANGELOG.md b/docs/i18n/pt-BR/CHANGELOG.md index 5db6d9b189e..2b69faea25f 100644 --- a/docs/i18n/pt-BR/CHANGELOG.md +++ b/docs/i18n/pt-BR/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/pt/CHANGELOG.md b/docs/i18n/pt/CHANGELOG.md index 364cc57c2e0..08bbf3db5ea 100644 --- a/docs/i18n/pt/CHANGELOG.md +++ b/docs/i18n/pt/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ro/CHANGELOG.md b/docs/i18n/ro/CHANGELOG.md index de3c1adc61b..a331e42c85a 100644 --- a/docs/i18n/ro/CHANGELOG.md +++ b/docs/i18n/ro/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ru/CHANGELOG.md b/docs/i18n/ru/CHANGELOG.md index e13ba3fb7a2..c2fa33e8b7b 100644 --- a/docs/i18n/ru/CHANGELOG.md +++ b/docs/i18n/ru/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/sk/CHANGELOG.md b/docs/i18n/sk/CHANGELOG.md index 40e3e3ae35b..74f75ba62cf 100644 --- a/docs/i18n/sk/CHANGELOG.md +++ b/docs/i18n/sk/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/sv/CHANGELOG.md b/docs/i18n/sv/CHANGELOG.md index 79e47d0c43c..c96db82ead4 100644 --- a/docs/i18n/sv/CHANGELOG.md +++ b/docs/i18n/sv/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/sw/CHANGELOG.md b/docs/i18n/sw/CHANGELOG.md index e0d05a0cb40..d9258be4c99 100644 --- a/docs/i18n/sw/CHANGELOG.md +++ b/docs/i18n/sw/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ta/CHANGELOG.md b/docs/i18n/ta/CHANGELOG.md index 60eb1fc2ce3..2a79a5512e4 100644 --- a/docs/i18n/ta/CHANGELOG.md +++ b/docs/i18n/ta/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/te/CHANGELOG.md b/docs/i18n/te/CHANGELOG.md index 78338abb5b1..62c87eba818 100644 --- a/docs/i18n/te/CHANGELOG.md +++ b/docs/i18n/te/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/th/CHANGELOG.md b/docs/i18n/th/CHANGELOG.md index c6aeb022bd9..ab7c3e44d91 100644 --- a/docs/i18n/th/CHANGELOG.md +++ b/docs/i18n/th/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/tr/CHANGELOG.md b/docs/i18n/tr/CHANGELOG.md index f39ee234763..67133c16578 100644 --- a/docs/i18n/tr/CHANGELOG.md +++ b/docs/i18n/tr/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/uk-UA/CHANGELOG.md b/docs/i18n/uk-UA/CHANGELOG.md index 258a01e20cb..9afca5cbc63 100644 --- a/docs/i18n/uk-UA/CHANGELOG.md +++ b/docs/i18n/uk-UA/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/ur/CHANGELOG.md b/docs/i18n/ur/CHANGELOG.md index c6834448a80..f740d738b2f 100644 --- a/docs/i18n/ur/CHANGELOG.md +++ b/docs/i18n/ur/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/vi/CHANGELOG.md b/docs/i18n/vi/CHANGELOG.md index ded47a11cfb..a0eb43f2a79 100644 --- a/docs/i18n/vi/CHANGELOG.md +++ b/docs/i18n/vi/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/i18n/zh-CN/CHANGELOG.md b/docs/i18n/zh-CN/CHANGELOG.md index 7dfd95db1eb..a2b4487077d 100644 --- a/docs/i18n/zh-CN/CHANGELOG.md +++ b/docs/i18n/zh-CN/CHANGELOG.md @@ -6,6 +6,14 @@ ## [3.8.25] — 2026-06-14 +## [3.8.26] — TBD + +_See English CHANGELOG for v3.8.26 details._ + +--- + +## [3.8.25] — 2026-06-14 + ### ✨ New Features - **feat(compression): pluggable compression engines + async pipeline + Compression Studios** — a new prompt-compression subsystem with selectable engines (Lite / Aggressive / Ultra), an asynchronous compression pipeline wired into the chat core, and "Compression Studios" tooling for inspecting and tuning compression. ([#3848](https://github.com/diegosouzapw/OmniRoute/pull/3848)) diff --git a/docs/ops/DOCUMENTATION_AUDIT_REPORT.md b/docs/ops/DOCUMENTATION_AUDIT_REPORT.md new file mode 100644 index 00000000000..767c97f3e24 --- /dev/null +++ b/docs/ops/DOCUMENTATION_AUDIT_REPORT.md @@ -0,0 +1,235 @@ +# OmniRoute — Relatório de Auditoria de Documentação & Plano de Sincronização + +> **Versão do projeto:** 3.8.24 · **Data:** 2026-06-13 · **Status:** FASE 1 (pesquisa/organização) — execução **pendente de confirmação** +> **Escopo:** docs raiz · `/docs` · site `:20128/docs` (Fumadocs) · Wiki GitHub · i18n (42 locales) · CI de docs/i18n + +--- + +## 0. TL;DR + +1. **As contagens estão dessincronizadas entre 5 fontes diferentes** (código, README, AGENTS.md, site, Wiki). O caso mais grave: **providers** aparece como `177` (README) / `232` (AGENTS) / `223` (gerador, **correto**) / `212+` (Wiki) / `160+` (CLAUDE.md). +2. **Os gates de CI atuais NÃO validam os números mais visíveis** (provider count, free count, test count, locale count). Por isso o drift passou despercebido. +3. **A Wiki do GitHub está órfã**: 995 páginas, **sem automação de sync**, último update genérico, números muito antigos (`14 strategies`, `37 MCP tools`, `212+ providers`). +4. **O pipeline i18n nunca rodou para os docs**: `.i18n-state.json` não existe → drift check não tem baseline. +5. **~10–12 funcionalidades recentes não têm documentação** (Plugin Marketplace, Free Provider Rankings/Arena ELO, IPv6 egress, Feature Flags, Notion/Obsidian context, etc.). + +--- + +## 1. Números canônicos REAIS (a fonte de verdade de cada um) + +| Métrica | **Valor real** | Fonte de verdade (como medir) | README | AGENTS.md | CLAUDE.md | docs/README.md | Wiki Home | Site | +| ------------------------------------- | ---------------------------------------------------- | --------------------------------------------------------------------------------- | -------------- | --------- | ---------- | --------------------------- | --------- | ----------------- | --- | +| **Providers (total)** | **223** | `scripts/docs/gen-provider-reference.ts` → `PROVIDER_REFERENCE.md` ("unique IDs") | ❌ 177 | ❌ 232 | ❌ "160+" | (n/a) | ❌ 212+ | via gerador | +| **Providers c/ free tier** | **103** `hasFree:true` / **98** pesquisados c/ quota | `grep hasFree:true providers.ts` / `FREE_TIERS.md` | ⚠️ "50+" | — | — | — | ⚠️ "50+" | — | +| **Free forever** | **11** (a revalidar) | README claim — sem fonte programática | "11" | — | — | — | — | — | +| **Test files (unit)** | **1.574** | `find tests/unit -name '*.test.ts'` | — | — | — | — | — | — | +| **Test files (integration)** | **76** | `find tests/integration` | — | — | — | — | — | — | +| **Test files (total)** | **~1.660** (+46 em src/open-sse) | find global | — | — | — | — | — | — | +| **Test cases (aprox)** | **~16.000** | `grep -E '(test | it)\(' tests/` | — | — | — | — | — | — | +| **`unit/` test files (CONTRIBUTING)** | **1.574** | — | — | — | — | ❌ **CONTRIBUTING diz 122** | — | — | +| **API endpoints (route.ts)** | **502** | `find src/app/api -name route.ts` | — | — | — | — | — | — | +| **Endpoints `/v1` (OpenAI-compat)** | **75** | `find src/app/api/v1 -name route.ts` | — | — | — | — | — | — | +| **MCP tools** | **87** (33 base + módulos) | `schemas/tools.ts` = 33 base; +memory/skill/notion/obsidian/gamification/plugin | ✅ 87 | ✅ 87 | ✅ 87 | — | ❌ 37 | — | +| **MCP scopes** | **30** (16 base em tools.ts) | `scopeEnforcement.ts` + módulos | — | ✅ 30 | ✅ 30 | — | — | — | +| **Routing strategies** | **15** | `open-sse/services/combo.ts` (gate valida) | ✅ 15 | ✅ 15 | ✅ 15 | ❌ 14 | ❌ 14 | — | +| **Auto-combo scoring factors** | **9** (label) / engine multifator | `AUTO-COMBO.md` | "9" | "12" | "9-factor" | ❌ "9-factor" | — | — | +| **i18n locales** | **42** (+en = 43) | `config/i18n.json` | — | ✅ 42 | — | ❌ 40 | ❌ "40+" | ✅ 40 (LANGUAGES) | +| **Executors** | **60** | gate valida ✓ | — | ✅ | — | — | — | — | +| **A2A skills** | **6** | gate valida ✓ | ✅ | ✅ | ✅ | — | — | — | +| **Cloud agents** | **3** | gate valida ✓ | ✅ | — | — | — | — | — | +| **OAuth flows / providers** | **16** flows / **19** providers OAuth | gate (16) vs `PROVIDER_REFERENCE` (19) | — | — | — | — | — | — | +| **DB modules / migrations** | **83 / 97** | gate/CLAUDE ✓ | — | ✅ | ✅ | — | — | — | + +> ⚠️ **Inconsistências de número que precisam de decisão de produto (não só correção mecânica):** +> +> - **Free count:** `hasFree:true` = 103, mas inclui créditos-de-cadastro one-time. `FREE_TIERS.md` documenta 98 pesquisados (≈63 recorrentes, 29 signup-only, 6 descontinuados). O "50+/11 forever" é conservador e defensável — **decidir o headline canônico**. +> - **Auto-combo "9-factor" vs "12":** README diz 9, AGENTS diz 12. Precisa alinhar à contagem real em `AUTO-COMBO.md`. + +--- + +## 2. Defasagens por fonte de documentação + +### 2.1 Documentos da raiz + +| Arquivo | Problema | Ação | +| -------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- | +| `README.md` | `177 providers` em ~9 lugares (linhas 9, 22, 23, 62, 139, 143, 266, 322, 326, 736) + badges + âncora `#-177-ai-providers--50-free` | Corrigir para **223**; revisar badge/âncora; revalidar "50+/11 forever" | +| `AGENTS.md` | `232 provider entries` (linha 6) + live counts `providers 232` (linha 11) | Corrigir para **223** | +| `CLAUDE.md` | `"160+"` providers (linha ~40) | Corrigir para **223** | +| `CONTRIBUTING.md` | `unit/ (122 test files)` (linha 255) — defasado em >1.400 | Corrigir para **1.574** (ou texto dinâmico) | +| `CHANGELOG.md` | OK (v3.8.24 correto) | — | +| `SECURITY.md` / `CODE_OF_CONDUCT.md` / `GEMINI.md` | Genéricos, OK | — | + +### 2.2 `/docs` + +| Arquivo | Problema | +| ---------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `docs/README.md` (índice) | `9-factor scoring, 14 strategies` (linha 81 → deveria ser 15); `40 locales` (linha 121 → 42) | +| `docs/guides/I18N.md` | "supports 30 languages" (real: 42); `lastUpdated 2026-05-13` | +| `docs/frameworks/AGENT-SKILLS.md, AGENTBRIDGE.md` | `lastUpdated 2026-05-28` (v3.8.6) | +| `docs/frameworks/WEBHOOKS.md`, `docs/guides/PWA_GUIDE.md` | `lastUpdated 2026-05-13` (v3.8.0) | +| `docs/frameworks/SEARCH_TOOLS_STUDIO.md`, `PLAYGROUND_STUDIO.md` | `lastUpdated 2026-05-30` | +| `docs/guides/TROUBLESHOOTING.md, FEATURES.md, UNINSTALL.md`, `docs/reference/API_REFERENCE.md` | refs a versões antigas (v3.5.x–v3.7.x) — verificar se históricas (ok) ou stale | +| Órfãos do site (existem em `/docs` mas fora do `meta.json`) | `routing/QUOTA_SHARE.md`, `guides/CODEX-CLI-CONFIGURATION.md`, `security/SOCKET_DEV_FINDINGS.md`, `compression/EXTENDING_COMPRESSION.md` — acessíveis por URL mas **não na sidebar** | +| Raiz `/docs` nunca no site | `AGENTROUTER.md`, `PROVIDERS.md`, `DOCUMENTATION_OVERHAUL_PLAN.md`, `SUBMIT_PR.md`, `fix-opencode-context.md` | + +### 2.3 Site `:20128/docs` (Fumadocs) + +- **Como funciona:** `docs//*.md` → `source.config.ts` (globs) → `.source/server.ts` (gerado) → `src/lib/source.ts` → `src/app/docs/layout.tsx` (sidebar = `pageTree` dos `meta.json`) → `[...slug]/page.tsx`. **60 docs em inglês** entram no site. +- **Navegação curada por `meta.json`** → arquivo novo em `/docs` **não aparece** até ser adicionado manualmente ao `meta.json` da seção. Hoje há 4 arquivos importados mas fora da sidebar (acima). +- **i18n no site:** `[...slug]/page.tsx` lê cookie `NEXT_LOCALE`; se ≠ en, tenta `docs/i18n//docs//.md` via `marked.parse()`, com fallback para o MDX inglês. Seletor: `LanguageSelector.tsx` (40 idiomas em `LANGUAGES`). +- **API Explorer:** `openapi.generated.ts` é gerado por `scripts/docs/gen-openapi-module.mjs` a partir de `docs/reference/openapi.yaml` no `prebuild:docs`. +- **Riscos de drift:** (a) `meta.json` manual; (b) traduções não atualizam quando o inglês muda; (c) `openapi.yaml` precisa de regen; (d) `LANGUAGES` no app diz 40, config diz 42 → **divergência app vs config**. + +### 2.4 Wiki do GitHub (`/wiki`) — **mais defasada de todas** + +- **995 páginas** (`60 docs Title-Case` + `935 i18n`), `Home.md`, `_Sidebar.md`, `_Footer.md`. +- **Sem nenhum script/automação de sync** no repo (`grep wiki` em `scripts/`, `.github/`, `package.json` = vazio). Foi populada uma vez, manualmente. +- **Números muito antigos no `Home.md`:** `212+ providers`, `14 Routing Strategies`, `MCP Server 37 tools`, `40+ Languages`. +- **Conclusão:** a Wiki não tem "fonte de verdade" — precisa virar **espelho automatizado** de `/docs` (ver Plano §6, Fase 4). + +### 2.5 i18n (42 locales) + +- **Fonte de verdade dos locales:** `config/i18n.json` → **42** (+en=43). Documentação diz 30 (`I18N.md`) e 40 (`docs/README.md`, `LANGUAGES` no app) — **3 números diferentes**. +- **Subset espelhado por idioma:** ~26–27 arquivos (7 raiz: README/CONTRIBUTING/CLAUDE/GEMINI/AGENTS/SECURITY/CODE_OF_CONDUCT + llm.txt/CHANGELOG copiados + ~19 docs em architecture/frameworks/guides/ops/reference/routing). +- **`.i18n-state.json` não existe** → `i18n:check` (drift) não tem baseline; tradução de docs nunca foi executada pelo pipeline novo. +- **Duplicações/legados:** `pt` vs `pt-BR` (ambos traduzem os mesmos arquivos); `id` vs `in` (Indonésio — `in` é legado ISO 639-3, deveria deprecar). +- **CLI locales incompletos:** `bn, gu, he, mr, ms, phi, in` = 3 bytes (vazios). +- **Motor:** `run-translation.mjs` usa endpoint OpenAI-compat via env `OMNIROUTE_TRANSLATION_*` (LLM); scripts Python (`i18n_autotranslate.py`, `generate-multilang.mjs`) marcados deprecated. + +--- + +## 3. Gaps de funcionalidades (features sem doc) — com curadoria + +> Curadoria aplicada: o agente de exploração marcou 57% dos módulos como "não documentados", mas muitos (`config`, `runtime`, `middleware`, `images`, `catalog`, `system`, `display`, `events`, `embeddings`) são **internos** e não merecem doc dedicado. Lista abaixo filtrada para o que é **voltado ao usuário/operador**. + +### P0 — features novas visíveis ao usuário, sem doc + +| Feature | Onde no código | PR | Doc sugerido | +| ------------------------------------------------------ | -------------------------------------------------------------------------- | ------------ | ---------------------------------------------------------- | +| **Plugin Marketplace** (customizável + SSRF hardening) | `src/app/api/plugins/marketplace/` | #3656, #3774 | `docs/frameworks/PLUGIN_MARKETPLACE.md` | +| **Free Provider Rankings (Arena ELO)** | `src/app/api/free-provider-rankings/`, `/dashboard/free-provider-rankings` | #3799 | `docs/guides/FREE_PROVIDER_RANKINGS.md` | +| **IPv6 egress family selector** (auto/ipv4/ipv6) | proxy/egress + UI form | #3777 | `docs/security/EGRESS_POLICY.md` (ou seção em PROXY_GUIDE) | +| **Feature Flags page (runtime + emergency fallback)** | `/dashboard` feature-flags | #3752, #3741 | `docs/reference/FEATURE_FLAGS.md` | + +### P1 — integrações/frameworks sem doc + +| Feature | Onde | Doc sugerido | +| -------------------------------------- | ----------------------------------- | ------------------------------------- | +| **Notion context source** | `src/lib/notion/` (+6 MCP tools) | `docs/frameworks/NOTION_CONTEXT.md` | +| **Obsidian context source** | `src/lib/obsidian/` (+22 MCP tools) | `docs/frameworks/OBSIDIAN_CONTEXT.md` | +| **Quota-shared routing audit** (#3779) | combo + quota | seção em `AUTO-COMBO.md` | +| **Model lockout / success-decay** | `RESILIENCE_GUIDE.md` desatualizado | atualizar `RESILIENCE_GUIDE.md` | +| **Cost/Spend tracking** | `/dashboard/costs` | `docs/guides/COST_TRACKING.md` | + +### P2 — sub-documentados + +Traffic Inspector, Search Tools Studio (raso), Prompt Caching, Credential Health, Background Jobs, Database Migrations guide. + +> **Validação obrigatória na execução:** cada item acima será confirmado no código (trust-but-verify) antes de escrever doc — não documentar feature que não exista/esteja como descrita. + +--- + +## 4. CI de docs/i18n — coberto vs lacunas + +### Coberto (hard gates) + +`check:docs-sync` (version package↔openapi↔CHANGELOG + mirrors i18n) · `check:env-doc-sync` (env code↔.env.example↔ENVIRONMENT.md) · `check:docs-symbols` (anti-alucinação rota) · `check:openapi-routes` · `check:cli-i18n` · `check-ui-keys-coverage` (floor 65%). + +### Parcial / advisory + +`check:docs-counts` (**soft, não no CI principal** — e **não cobre providers/free/tests/locales**) · `check:doc-links` (só internos) · `check:fabricated-docs` (soft) · `check-translation-drift` (`--warn`, não bloqueia) · `validate_translation.py` (matrix `continue-on-error`). + +### Lacunas (sem gate algum) + +1. **Provider count / free count / test count / locale count** — os números mais visíveis **não são validados**. → causa-raiz de todo o drift atual. +2. **Validação MDX/Fumadocs** — quebra de sintaxe só aparece no deploy. +3. **Lint de prosa (Vale/markdownlint)** — sem checagem de estilo/ortografia. +4. **Links externos** — `check:doc-links` ignora http(s); URLs mortas passam. +5. **Imagens/screenshots/diagramas órfãos.** +6. **Drift de tradução não é blocking.** +7. **`meta.json` ↔ `/docs`** — arquivo novo fora da sidebar não é detectado. +8. **Wiki sync** — inexistente. +9. **`config/i18n.json` (42) vs `LANGUAGES` app (40)** — sem gate de consistência. + +--- + +## 5. Boas práticas 2026 (pesquisa web) aplicáveis + +| Prática | Ferramenta | Aplicação no OmniRoute | +| ----------------------------------- | -------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------- | +| **Lint de prosa em CI** | **Vale** (Google/Microsoft style) + **markdownlint** | Novo job `docs-lint` (warning-first p/ não travar) | +| **Severidade graduada** | error = link quebrado / code-fence / alt-text faltando; warning = voz passiva / estilo | Configurar `.vale.ini` + `.markdownlint.json` | +| **Feedback local rápido** | pre-commit com markdownlint/Vale (<2s) | Adicionar ao husky lint-staged p/ `*.md` | +| **Accept-list de vocabulário** | `Vale accept.txt` | Evitar ruído com termos do projeto (OmniRoute, combo, etc.) | +| **Link checker robusto** | **lychee** (Rust, externos + internos, cache) | Job semanal/cron + flag opcional no doc-links | +| **Wiki como espelho automatizado** | **`Andrew-Chen-Wang/github-wiki-action`** ou `wiki-sync` | Workflow que espelha `/docs` → `.wiki.git` em push to main | +| **Tradução LLM roteada por tarefa** | docs técnicos → GPT-5.x; nuance → Claude; bulk → modelo barato | Já temos roteamento próprio — usar `cx/gpt-5.4-mini` p/ docs (config existente) | +| **Translation memory + glossário** | reduz drift e protege termos | Adotar glossário/accept-list compartilhado UI+docs | +| **TMS via MCP** | Crowdin/Lokalise/Tolgee/SimpleLocalize têm MCP oficial | Opcional futuro; hoje pipeline próprio já cobre | +| **Gerar docs no CI** | docs sempre refletem o código | Elevar gerador de provider/openapi + count-guard a gate | + +**Fontes:** [Fern — Docs Linting Guide (jan/2026)](https://buildwithfern.com/post/docs-linting-guide) · [Netlify — Docs Linting in CI/CD](https://www.netlify.com/blog/a-key-to-high-quality-documentation-docs-linting-in-ci-cd/) · [GitLab Docs — Documentation testing](https://docs.gitlab.com/development/documentation/testing/) · [Lokalise — Best LLM for translation 2026](https://lokalise.com/blog/what-is-the-best-llm-for-translation/) · [Crowdin — AI Localization 2026](https://crowdin.com/blog/ai-localization) · [Andrew-Chen-Wang/github-wiki-action](https://github.com/Andrew-Chen-Wang/github-wiki-action) · [OneUptime — Generate Docs with GitHub Actions](https://oneuptime.com/blog/post/2026-01-27-generate-documentation-github-actions/view) + +--- + +## 6. PLANO DE MELHORIAS & SINCRONIZAÇÃO (execução pós-confirmação) + +### Fase A — Números canônicos (correção mecânica de alto impacto) + +1. Regenerar `PROVIDER_REFERENCE.md` (`gen-provider-reference.ts`) e fixar **223** como fonte. +2. Corrigir **providers** em: README (9 ocorrências + badge + âncora), AGENTS.md, CLAUDE.md, Wiki Home → **223**. +3. Corrigir **tests** em CONTRIBUTING.md (122 → 1.574) — ou tornar texto dinâmico. +4. Corrigir **strategies** (14 → 15) e **locales** (30/40 → 42) em `docs/README.md`, `I18N.md`, Wiki Home, e `LANGUAGES` do app. +5. Corrigir **MCP tools** na Wiki (37 → 87) e alinhar **auto-combo factors** (9 vs 12 → valor real). +6. Decidir headline **free** (50+/11 vs 98/103) e aplicar uniformemente. + +### Fase B — Sincronizar /docs + README (estrutural) + +7. Atualizar `lastUpdated`/versão dos docs defasados (AGENT-SKILLS, WEBHOOKS, PWA, I18N, SEARCH/PLAYGROUND_STUDIO). +8. Adicionar os 4 arquivos órfãos ao `meta.json` (ou removê-los conscientemente). +9. Avaliar README: adicionar/atualizar seções (Quick Start, tabela de números, links p/ novos docs). + +### Fase C — Novos documentos (gaps de features, P0→P1) + +10. Criar P0: PLUGIN_MARKETPLACE, FREE_PROVIDER_RANKINGS, EGRESS_POLICY, FEATURE_FLAGS. +11. Atualizar RESILIENCE_GUIDE (model lockout) e AUTO-COMBO (quota-shared). +12. Criar P1: NOTION_CONTEXT, OBSIDIAN_CONTEXT, COST_TRACKING (conforme confirmação). + +### Fase D — i18n + +13. Bootstrapar `.i18n-state.json` (`i18n:run --dry-run`) e rodar tradução dos docs corrigidos. +14. Reconciliar `config/i18n.json` (42) ↔ `LANGUAGES` app (40); decidir sobre `in` (deprecar) e `pt`/`pt-BR`. +15. Atualizar I18N.md com o processo real e contagem 42. + +### Fase E — Site `:20128/docs` + +16. Regenerar `openapi.generated.ts` e validar API Explorer. +17. Garantir que os novos docs entram no `meta.json` e renderizam (verificação visual via browser). + +### Fase F — Wiki GitHub (automatizar) + +18. Criar workflow `wiki-sync.yml` espelhando `/docs` → `.wiki.git` (github-wiki-action) — **fim do drift manual**. +19. Re-sincronizar a Wiki com os números corrigidos. + +### Fase G — CI de docs (fechar lacunas) + +20. **Adicionar count-guard** a `check:docs-counts`: providers, free, tests, locales, MCP tools/scopes → **gate blocking** (matando a causa-raiz). +21. Promover `check-translation-drift` a blocking (`--strict`) após baseline. +22. Adicionar job advisory `docs-lint` (Vale + markdownlint) e link-check externo (lychee, cron). +23. Adicionar gate de consistência `config/i18n.json ↔ LANGUAGES`. + +--- + +## 7. Decisões necessárias do usuário (antes de executar) + +1. **Headline de "free"**: manter `50+ / 11 forever` ou adotar número pesquisado (`98` documentados)? +2. **Escopo dos novos docs**: criar todos P0+P1 agora, ou só P0 nesta rodada? +3. **Wiki**: automatizar via workflow (recomendado) ou só re-sincronizar manualmente desta vez? +4. **i18n**: re-traduzir os docs alterados agora (custa chamadas LLM) ou só corrigir o inglês e deixar i18n para um passo seguinte? +5. **`in`/`pt` legados**: deprecar `in` (Indonésio legado) nesta rodada? +6. **CI**: implementar os novos gates (count-guard, Vale, wiki-sync) nesta rodada ou em PR separado? + +--- + +_Relatório gerado na Fase 1 (pesquisa). Nenhum documento de produto foi alterado ainda. A sincronização começa após confirmação do escopo acima._ diff --git a/docs/ops/meta.json b/docs/ops/meta.json index b2248086f10..dfc1af0180f 100644 --- a/docs/ops/meta.json +++ b/docs/ops/meta.json @@ -8,6 +8,7 @@ "PROXY_GUIDE", "SQLITE_RUNTIME", "COVERAGE_PLAN", - "E2E_DASHBOARD_SHAKEDOWN_v3.8.0" + "E2E_DASHBOARD_SHAKEDOWN_v3.8.0", + "DOCUMENTATION_AUDIT_REPORT" ] } diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index 7170e7c15fd..0929accccf0 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -1,18 +1,16 @@ --- title: "Provider Reference" -version: 3.8.12 -lastUpdated: 2026-06-06 +version: 3.8.25 +lastUpdated: 2026-06-15 --- # Provider Reference -> **For Users**: Looking for a simple guide? See the [Providers Guide](../getting-started/PROVIDERS-GUIDE.md) for step-by-step instructions. - > **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. > Regenerate with: `npm run gen:provider-reference` -> **Last generated:** 2026-06-06 +> **Last generated:** 2026-06-15 -Total providers: **223**. See category breakdown below. +Total providers: **226**. See category breakdown below. ## Categories @@ -35,273 +33,276 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each ## OAuth Providers (19) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | -| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | -| `antigravity` | — | Antigravity | OAuth | — | — | -| `claude` | `cc` | Claude Code | OAuth | — | — | -| `cline` | `cl` | Cline | OAuth | — | — | -| `codex` | `cx` | OpenAI Codex | OAuth | — | — | -| `cursor` | `cu` | Cursor IDE | OAuth | — | — | -| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | -| `gemini-cli` | `gemini-cli` | Gemini CLI | OAuth | — | Uses Gemini CLI OAuth / Cloud Code credentials. Pro models require an eligible Google account or paid plan. | -| `github` | `gh` | GitHub Copilot | OAuth | — | — | -| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | OAuth application with ai_features + read_user scopes. Configure GITLAB_DUO_OAUTH_CLIENT_ID and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET on this OmniRoute instance. | -| `kilocode` | `kc` | Kilo Code | OAuth | — | — | -| `kimi-coding` | `kmc` | Kimi Coding | OAuth | — | — | -| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | -| `qoder` | `if` | Qoder AI | OAuth | — | — | -| `qwen` | `qw` | Qwen Code | OAuth | — | ⚠️ **DEPRECATED.** Qwen OAuth free tier was discontinued on 2026-04-15. Use 'bailian-coding-plan', 'alibaba', 'alibaba-cn', or 'openrouter' provider with API key instead. | -| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | -| `windsurf` | `ws` | Windsurf (Devin CLI) | OAuth | [link](https://windsurf.com) | Sign in at windsurf.com to get your token. Visit windsurf.com/show-auth-token after logging in and paste it here, or use the device-code login flow. | -| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | +| ID | Alias | Name | Tags | Website | Notes | +| ------------- | ------------ | -------------------- | ----- | ------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | +| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | +| `antigravity` | — | Antigravity | OAuth | — | — | +| `claude` | `cc` | Claude Code | OAuth | — | — | +| `cline` | `cl` | Cline | OAuth | — | — | +| `codex` | `cx` | OpenAI Codex | OAuth | — | — | +| `cursor` | `cu` | Cursor IDE | OAuth | — | — | +| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | +| `gemini-cli` | `gemini-cli` | Gemini CLI | OAuth | — | Uses Gemini CLI OAuth / Cloud Code credentials. Pro models require an eligible Google account or paid plan. | +| `github` | `gh` | GitHub Copilot | OAuth | — | — | +| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | OAuth application with ai_features + read_user scopes. Configure GITLAB_DUO_OAUTH_CLIENT_ID and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET on this OmniRoute instance. | +| `kilocode` | `kc` | Kilo Code | OAuth | — | — | +| `kimi-coding` | `kmc` | Kimi Coding | OAuth | — | — | +| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | +| `qoder` | `if` | Qoder AI | OAuth | — | — | +| `qwen` | `qw` | Qwen Code | OAuth | — | ⚠️ **DEPRECATED.** Qwen OAuth free tier was discontinued on 2026-04-15. Use 'bailian-coding-plan', 'alibaba', 'alibaba-cn', or 'openrouter' provider with API key instead. | +| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | +| `windsurf` | `ws` | Windsurf (Devin CLI) | OAuth | [link](https://windsurf.com) | In the Windsurf / VS Code IDE, open the command palette and run `Windsurf: Provide Auth Token` (or click the Jupyter "Get Windsurf Authentication Token" button), then copy the shown token and paste it here. Note: opening windsurf.com/show-auth-token directly only renders a "Redirecting" page — the IDE must initiate the flow (it adds a `?state=...` param) for the token to appear. | +| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | -## Web Cookie Providers (20) +## Web Cookie Providers (22) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | -| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | -| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | -| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | -| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste your access_token from copilot.microsoft.com (or export a .har file from DevTools while logged in) | -| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | -| `doubao-web` | `db` | Doubao Web (ByteDance) | Web cookie | [link](https://www.doubao.com) | Paste your session cookie from doubao.com (DevTools → Application → Cookies) | -| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | -| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | -| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste your hf-chat cookie value from huggingface.co/chat (DevTools → Application → Cookies → hf-chat). Optional — works without auth for basic use. | -| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | -| `kimi-web` | `kimi-web` | Kimi Web (Moonshot AI) | Web cookie | [link](https://kimi.moonshot.cn) | Paste your session cookie from kimi.moonshot.cn (DevTools → Application → Cookies) | -| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your abra_sess value or full cookie header from meta.ai | -| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | -| `phind` | `ph` | Phind (Free) | Web cookie | [link](https://www.phind.com) | Paste your session cookie from phind.com (DevTools → Application → Cookies). Optional — works with free tier. | -| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | -| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | -| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | -| `v0-vercel-web` | `v0` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | -| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | +| ID | Alias | Name | Tags | Website | Notes | +| ----------------- | ------------- | ---------------------------- | ---------- | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your \_\_client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | +| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your \_\_Secure-authjs.session-token value or full cookie header from app.blackbox.ai | +| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your \_\_Secure-next-auth.session-token cookie value from chatgpt.com | +| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | +| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste your access_token from copilot.microsoft.com (or export a .har file from DevTools while logged in) | +| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | +| `doubao-web` | `db` | Doubao Web (ByteDance) | Web cookie | [link](https://www.doubao.com) | Paste your session cookie from doubao.com (DevTools → Application → Cookies) | +| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy **Secure-1PSID and **Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | +| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your **Secure-1PSID cookie value from gemini.google.com. Optionally add **Secure-1PSIDTS separated by semicolon. | +| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | +| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste your hf-chat cookie value from huggingface.co/chat (DevTools → Application → Cookies → hf-chat). Optional — works without auth for basic use. | +| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | +| `kimi-web` | `kimi-web` | Kimi Web (Moonshot AI) | Web cookie | [link](https://kimi.moonshot.cn) | Paste your session cookie from kimi.moonshot.cn (DevTools → Application → Cookies) | +| `lmarena` | `lma` | LMArena (Free) | Web cookie | [link](https://lmarena.ai) | Paste your session cookie from lmarena.ai (DevTools → Application → Cookies). Optional — works with free tier for basic comparisons. | +| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your abra_sess value or full cookie header from meta.ai | +| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your \_\_Secure-next-auth.session-token cookie value from perplexity.ai | +| `phind` | `ph` | Phind (Free) | Web cookie | [link](https://www.phind.com) | Paste your session cookie from phind.com (DevTools → Application → Cookies). Optional — works with free tier. | +| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | +| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | +| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | +| `v0-vercel-web` | `v0` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | +| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | -## API Key Providers (paid / paid-with-free-credits) (151) +## API Key Providers (paid / paid-with-free-credits) (152) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | -| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | -| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | -| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | $0.025/day free credits — 200+ models (GPT-4o, Claude, Gemini, Llama) via single endpoint | -| `alibaba` | `ali` | Alibaba | API key | [link](https://dashscope-intl.aliyuncs.com) | — | -| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.aliyuncs.com) | — | -| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | -| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | -| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | -| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | -| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | -| `baichuan` | `baichuan` | Baichuan | API key | [link](https://baichuan.com) | Get API key at platform.baichuan-ai.com | -| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://yiyan.baidu.com) | Get API key at console.bce.baidu.com | -| `bailian-coding-plan` | `bcp` | Alibaba Coding Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/coding-plan) | — | -| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | -| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Free tier with auto:free routing — zero-cost inference, no credit card required | -| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | -| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | -| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Free tier: unlimited basic chat plus Minimax-M2.5, no credit card required | -| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | -| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | -| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | -| `cablyai` | `cablyai` | CablyAI | API key, aggregator | [link](https://cablyai.com) | Bearer API key for the CablyAI OpenAI-compatible gateway. | -| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | -| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | -| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | -| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | -| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | -| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | -| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | -| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | -| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | -| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | -| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | -| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | -| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | -| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | -| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | -| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | -| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | -| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | -| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | -| `firecrawl` | `fc` | Firecrawl | API key | [link](https://firecrawl.dev) | — | -| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | -| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | -| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | -| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | -| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | — | -| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com | -| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | -| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | -| `github-models` | `ghm` | GitHub Models | API key | [link](https://github.com/marketplace/models) | Create a GitHub PAT with 'models: read' scope at github.com/settings/tokens | -| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | -| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required | -| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required | -| `glhf` | `glhf` | GLHF Chat | API key, aggregator | [link](https://glhf.chat) | Bearer API key for the GLHF OpenAI-compatible gateway. | -| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | -| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | -| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | -| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | -| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | -| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | -| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | -| `huggingchat` | `huggingchat` | HuggingChat | API key | [link](https://huggingface.co/chat) | No API key required for basic access. | -| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | -| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | -| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | -| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | -| `inclusionai` | `inclusion` | InclusionAI | API key | [link](https://inclusionai.com) | Get API key at inclusionai.com | -| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | -| `jina-ai` | `jina` | Jina AI | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for the Jina AI rerank API. | -| `jina-reader` | `jr` | Jina Reader | API key | [link](https://jina.ai/reader) | — | -| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | -| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | -| `kimi` | `kimi` | Kimi | API key | [link](https://platform.moonshot.ai) | — | -| `kimi-coding-apikey` | `kmca` | Kimi Coding (API Key) | API key | [link](https://www.kimi.com/code) | — | -| `kluster` | `kluster` | Kluster AI | API key | [link](https://kluster.ai) | $5 free credits on signup - DeepSeek R1, Llama 4 Maverick/Scout, Qwen3 235B | -| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | -| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | -| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | -| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | -| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | -| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | No signup required - 2 req/s, 20 RPM, 100 req/hr free tier | -| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: 5M tokens/day on LongCat-2.0-Preview (Flash models retired 2026-05-29); up to 120M/day via feedback. | -| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | -| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | -| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | -| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | -| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | -| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | -| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | Get API key at monsterapi.ai | -| `moonshot` | `moonshot` | Moonshot AI | API key | [link](https://platform.moonshot.ai) | — | -| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | -| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | -| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | -| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | -| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | -| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | -| `novita` | `novita` | Novita AI | API key, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | -| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | -| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | -| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | -| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/api-keys) | — | -| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | -| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | -| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | -| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | -| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | -| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | -| `phind` | `phind` | Phind | API key | [link](https://phind.com) | Get API key at phind.com | -| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | -| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | -| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | No API key required for free public endpoint. Optional Spore tier: ~0.01 pollen/hour. | -| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | $25 free trial credits (30-day validity) | -| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Free community inference tier for testing | -| `puter` | `pu` | Puter AI | API key | [link](https://puter.com) | Get token at puter.com/dashboard → Copy Auth Token | -| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product/wenxinworkshop) | — | -| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | -| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | -| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | -| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | -| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | -| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/ai/generative-apis) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | -| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | -| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus permanently free models after identity verification | -| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | -| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | -| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | -| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | -| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | -| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | -| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | -| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | -| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | $25 signup credits + 3 permanently free models: Llama 3.3 70B, Vision, DeepSeek-R1 distill | -| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | -| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | -| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. | -| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | -| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | -| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | -| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | -| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | -| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | -| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | -| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | -| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | -| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | -| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | — | -| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | -| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | -| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | +| ID | Alias | Name | Tags | Website | Notes | +| --------------------- | -------------- | ------------------------------- | --------------------- | -------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | +| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | +| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | +| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | $0.025/day free credits — 200+ models (GPT-4o, Claude, Gemini, Llama) via single endpoint | +| `alibaba` | `ali` | Alibaba | API key | [link](https://dashscope-intl.aliyuncs.com) | — | +| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.aliyuncs.com) | — | +| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | +| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | +| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | +| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | +| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | +| `baichuan` | `baichuan` | Baichuan | API key | [link](https://baichuan.com) | Get API key at platform.baichuan-ai.com | +| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://yiyan.baidu.com) | Get API key at console.bce.baidu.com | +| `bailian-coding-plan` | `bcp` | Alibaba Coding Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/coding-plan) | — | +| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | +| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Free tier with auto:free routing — zero-cost inference, no credit card required | +| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | +| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Free tier: unlimited basic chat plus Minimax-M2.5, no credit card required | +| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | +| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | +| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | +| `cablyai` | `cablyai` | CablyAI | API key, aggregator | [link](https://cablyai.com) | Bearer API key for the CablyAI OpenAI-compatible gateway. | +| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | +| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | +| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | +| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | +| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | +| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | +| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | +| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | +| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | +| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | +| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | +| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | +| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | +| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | +| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | +| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | +| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | +| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | +| `firecrawl` | `fc` | Firecrawl | API key | [link](https://firecrawl.dev) | — | +| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | +| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | +| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | +| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | +| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | — | +| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com | +| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | +| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | +| `github-models` | `ghm` | GitHub Models | API key | [link](https://github.com/marketplace/models) | Create a GitHub PAT with 'models: read' scope at github.com/settings/tokens | +| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | +| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required | +| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free tier available — no credit card required | +| `glhf` | `glhf` | GLHF Chat | API key, aggregator | [link](https://glhf.chat) | Bearer API key for the GLHF OpenAI-compatible gateway. | +| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | +| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | +| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | +| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | +| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | +| `huggingchat` | `huggingchat` | HuggingChat | API key | [link](https://huggingface.co/chat) | No API key required for basic access. | +| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | +| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | +| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | +| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `inclusionai` | `inclusion` | InclusionAI | API key | [link](https://inclusionai.com) | Get API key at inclusionai.com | +| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | +| `jina-ai` | `jina` | Jina AI | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for the Jina AI rerank API. | +| `jina-reader` | `jr` | Jina Reader | API key | [link](https://jina.ai/reader) | — | +| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | +| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | +| `kimi` | `kimi` | Kimi | API key | [link](https://platform.moonshot.ai) | — | +| `kimi-coding-apikey` | `kmca` | Kimi Coding (API Key) | API key | [link](https://www.kimi.com/code) | — | +| `kluster` | `kluster` | Kluster AI | API key | [link](https://kluster.ai) | $5 free credits on signup - DeepSeek R1, Llama 4 Maverick/Scout, Qwen3 235B | +| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | +| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | +| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | +| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | +| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | +| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | No signup required - 2 req/s, 20 RPM, 100 req/hr free tier | +| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: 5M tokens/day on LongCat-2.0-Preview (Flash models retired 2026-05-29); up to 120M/day via feedback. | +| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | +| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | +| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | +| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | +| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | +| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | +| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | Get API key at monsterapi.ai | +| `moonshot` | `moonshot` | Moonshot AI | API key | [link](https://platform.moonshot.ai) | — | +| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | +| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | +| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | +| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | +| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | +| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | +| `novita` | `novita` | Novita AI | API key, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | +| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | +| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | +| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | +| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/api-keys) | — | +| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | +| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | +| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | +| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | +| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | +| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | +| `phind` | `phind` | Phind | API key | [link](https://phind.com) | Get API key at phind.com | +| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | +| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | +| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | No API key required for free public endpoint. Optional Spore tier: ~0.01 pollen/hour. | +| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | $25 free trial credits (30-day validity) | +| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid | +| `puter` | `pu` | Puter AI | API key | [link](https://puter.com) | Get token at puter.com/dashboard → Copy Auth Token | +| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product/wenxinworkshop) | — | +| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | +| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | +| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | +| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | +| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | +| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/ai/generative-apis) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | +| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | +| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus permanently free models after identity verification | +| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | +| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | +| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | +| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | +| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | +| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | +| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | +| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | $25 signup credits + 3 permanently free models: Llama 3.3 70B, Vision, DeepSeek-R1 distill | +| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | +| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. | +| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | +| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | +| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | +| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | +| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | +| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | +| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | +| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | +| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | +| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | +| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | — | +| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | +| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | +| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | +| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. | ## Local Providers (11) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | -| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | -| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | -| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | -| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | -| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | -| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | -| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | -| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | -| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | -| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | +| ID | Alias | Name | Tags | Website | Notes | +| --------------------- | ------------ | ------------------- | ------------------ | --------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | +| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | +| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | +| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | +| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | +| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | +| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | +| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | +| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | ## Search Providers (11) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | -| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | -| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | -| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | -| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/api-keys) | Same API key as Ollama Cloud (from ollama.com/settings/api-keys) | -| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | -| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs) | API key from SearchAPI (query param or Bearer auth) | -| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | -| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | -| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | -| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/docs/search/overview) | X-API-Key from the You.com platform dashboard | +| ID | Alias | Name | Tags | Website | Notes | +| ------------------- | --------------- | -------------------------- | ------ | --------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | +| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | +| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | +| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | +| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | +| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/api-keys) | Same API key as Ollama Cloud (from ollama.com/settings/api-keys) | +| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | +| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs) | API key from SearchAPI (query param or Bearer auth) | +| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | +| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | +| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | +| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/docs/search/overview) | X-API-Key from the You.com platform dashboard | ## Audio-only Providers (7) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | -| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | -| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | -| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | -| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | -| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | -| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | +| ID | Alias | Name | Tags | Website | Notes | +| ------------ | ---------- | ---------- | ----- | ------------------------------------- | ----------------------------------------------------------------------------------------------- | +| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | +| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | +| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | +| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | +| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | +| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | +| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | ## Upstream Proxy Providers (2) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | -| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | +| ID | Alias | Name | Tags | Website | Notes | +| ------------- | ----- | ----------- | -------------- | ---------------------------------------------------- | ----- | +| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | +| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | ## Cloud Agent Providers (3) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | -| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | -| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | +| ID | Alias | Name | Tags | Website | Notes | +| ------------- | ------------- | ------------ | ----------- | -------------------------------- | ----------------------------------------------------------- | +| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | +| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | +| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | ## System Providers (1) -| ID | Alias | Name | Tags | Website | Notes | -|----|-------|------|------|---------|-------| -| `auto` | `auto` | Auto (Zero-Config) | System | — | — | +| ID | Alias | Name | Tags | Website | Notes | +| ------ | ------ | ------------------ | ------ | ------- | ----- | +| `auto` | `auto` | Auto (Zero-Config) | System | — | — | ## Sources of truth diff --git a/docs/reference/openapi.yaml b/docs/reference/openapi.yaml index 2a4c081486d..102fef12b00 100644 --- a/docs/reference/openapi.yaml +++ b/docs/reference/openapi.yaml @@ -1,7 +1,7 @@ openapi: 3.1.0 info: title: OmniRoute API - version: 3.8.25 + version: 3.8.26 description: | OmniRoute is a local-first AI API proxy router. It provides an OpenAI-compatible endpoint that routes requests to multiple AI providers with load balancing, diff --git a/electron/package-lock.json b/electron/package-lock.json index 4adfd09b33e..9a2e3e1b74e 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -1,12 +1,12 @@ { "name": "omniroute-desktop", - "version": "3.8.25", + "version": "3.8.26", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "omniroute-desktop", - "version": "3.8.25", + "version": "3.8.26", "license": "MIT", "dependencies": { "electron-updater": "^6.8.9" diff --git a/electron/package.json b/electron/package.json index a4346545921..9a4b6b138b6 100644 --- a/electron/package.json +++ b/electron/package.json @@ -1,6 +1,6 @@ { "name": "omniroute-desktop", - "version": "3.8.25", + "version": "3.8.26", "description": "OmniRoute Desktop Application", "main": "main.js", "author": { diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 26b0e833a8a..e18b5faa47c 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -16,6 +16,30 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ }); export const GLM_SHARED_MODELS = Object.freeze([ + { + id: "glm-5.2", + name: "GLM 5.2", + contextLength: 1000000, + maxOutputTokens: 131072, + toolCalling: true, + supportsReasoning: true, + }, + { + id: "glm-5.2-high", + name: "GLM 5.2 High", + contextLength: 1000000, + maxOutputTokens: 131072, + toolCalling: true, + supportsReasoning: true, + }, + { + id: "glm-5.2-max", + name: "GLM 5.2 Max", + contextLength: 1000000, + maxOutputTokens: 131072, + toolCalling: true, + supportsReasoning: true, + }, { id: "glm-5.1", name: "GLM 5.1", diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts index 5dfa8495d90..07311134d3c 100644 --- a/open-sse/config/providerRegistry.ts +++ b/open-sse/config/providerRegistry.ts @@ -4535,6 +4535,11 @@ export function generateModels(): Record { if (!models[key]) { models[key] = entry.models; } + // Also store under the raw provider id so getProviderModels(id) works + // even when the provider has a different alias (e.g. "github" → alias "gh"). + if (entry.alias && entry.alias !== entry.id && !models[entry.id]) { + models[entry.id] = entry.models; + } } } return models; diff --git a/open-sse/executors/default.ts b/open-sse/executors/default.ts index 2ec6fbd6b3d..a585c0a72f2 100644 --- a/open-sse/executors/default.ts +++ b/open-sse/executors/default.ts @@ -151,6 +151,14 @@ function normalizeOpenAIChatUrl(baseUrl) { return `${normalized}/v1/chat/completions`; } +function getOpenRouterConnectionPreset( + providerSpecificData?: Record | null +): string | null { + const preset = + typeof providerSpecificData?.preset === "string" ? providerSpecificData.preset.trim() : ""; + return preset || null; +} + export class DefaultExecutor extends BaseExecutor { constructor(provider) { super(provider, PROVIDERS[provider] || PROVIDERS.openai); @@ -564,6 +572,16 @@ export class DefaultExecutor extends BaseExecutor { } } } + + if (this.provider === "openrouter") { + const connectionPreset = getOpenRouterConnectionPreset(credentials?.providerSpecificData); + if (connectionPreset && (withDefaults as Record).preset === undefined) { + withDefaults = { + ...(withDefaults as Record), + preset: connectionPreset, + }; + } + } } if (this.provider === "qwen" && typeof withDefaults === "object" && withDefaults !== null) { diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 6e78ead8db8..f83c6ae260c 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -50,6 +50,19 @@ function getEffectiveKey(credentials: ProviderCredentials): string { return credentials.apiKey || credentials.accessToken || ""; } +/** + * GLM-5.2 effort tiers route exclusively through the Anthropic transport, + * where Zhipu maps Claude Code effort selectors (high/max) to reasoning + * intensity. The base model ID sent upstream is always "glm-5.2". + * + * https://docs.z.ai/devpack/latest-model + */ +function parseGlm52Effort(model: string): { baseModel: string; effort: "high" | "max" } | null { + if (model === "glm-5.2-high") return { baseModel: "glm-5.2", effort: "high" }; + if (model === "glm-5.2-max") return { baseModel: "glm-5.2", effort: "max" }; + return null; +} + function applyGlmRequestDefaults(body: unknown, defaults?: JsonRecord | null): unknown { const record = asRecord(body); if (!record || !defaults) return body; @@ -228,27 +241,61 @@ export class GlmExecutor extends DefaultExecutor { credentials: ProviderCredentials, transport: GlmTransport ) { - const transformed = this.transformRequest(model, body, stream, credentials); + const effortTier = parseGlm52Effort(model); + const effectiveModel = effortTier ? effortTier.baseModel : model; + + const transformed = this.transformRequest(effectiveModel, body, stream, credentials); + const record = asRecord(transformed); + + // Ensure upstream receives the base model ID, not the effort-suffixed alias + if (record && effortTier) { + record.model = effectiveModel; + } if (transport === "openai") { - const record = asRecord(transformed); if (record && stream && hasTools(record) && record.tool_stream === undefined) { return { ...record, tool_stream: true }; } return transformed; } - return translateRequest( + const translated = translateRequest( FORMATS.OPENAI, FORMATS.CLAUDE, - model, - { ...(transformed as JsonRecord), _disableToolPrefix: true }, + effectiveModel, + { ...(record ?? {}), _disableToolPrefix: true }, stream, credentials, this.provider, null, { preserveCacheControl: false } ); + + // Inject effort and thinking for the Anthropic transport. + // Zhipu's Anthropic endpoint requires thinking.type=enabled to emit + // thinking_delta blocks in the SSE response. Without it, reasoning is + // not surfaced and clients see no thinking content. + // The effort-2025-11-24 beta header (in GLM_ANTHROPIC_BETA) carries + // the high/max intensity selector. + if (effortTier) { + const translatedRecord = asRecord(translated); + if (translatedRecord) { + translatedRecord.effort = effortTier.effort; + // Zhipu's Anthropic endpoint only supports thinking.type + // "enabled"/"disabled" — not "adaptive". Clients like Claude Code + // default to "adaptive" for reasoning models, so force "enabled" + // here while preserving any other fields (e.g. budget_tokens). + const existingThinking = asRecord(translatedRecord.thinking); + if (!existingThinking || existingThinking.type !== "enabled") { + translatedRecord.thinking = { + ...existingThinking, + type: "enabled", + }; + } + } + } + + return translated; } private async executeTransport( @@ -343,6 +390,15 @@ export class GlmExecutor extends DefaultExecutor { } async execute(input: ExecuteInput): Promise { + const effortTier = parseGlm52Effort(input.model); + + // GLM-5.2 effort tiers route directly through Anthropic transport (no fallback). + // Zhipu only graduates effort on the Anthropic endpoint via the + // effort-2025-11-24 beta header included in GLM_ANTHROPIC_BETA. + if (effortTier) { + return this.executeTransport(input, "anthropic"); + } + const primaryTransport = getGlmTransport( input.credentials.providerSpecificData, this.config.baseUrl diff --git a/open-sse/mcp-server/__tests__/audit.test.ts b/open-sse/mcp-server/__tests__/audit.test.ts index 90e5713a589..d51541f17f4 100644 --- a/open-sse/mcp-server/__tests__/audit.test.ts +++ b/open-sse/mcp-server/__tests__/audit.test.ts @@ -86,4 +86,57 @@ describe("MCP audit shutdown", () => { expect(audit.closeAuditDb()).toBe(true); expect(mockDb.close).toHaveBeenCalledTimes(1); }); + + it("falls back to node:sqlite when better-sqlite3 binding is missing", async () => { + const [maj, min] = process.versions.node.split(".").map(Number); + if (maj < 22 || (maj === 22 && min < 5)) { + return; // node:sqlite not available on this Node, skip + } + + // Simulate a global-install scenario where the bundled native binary + // never landed in dist/node_modules/better-sqlite3/build/Release/. + const bindingErr = new Error( + "Could not locate the bindings file. Tried: …/better_sqlite3.node" + ) as Error & { code?: string }; + bindingErr.code = "MODULE_NOT_FOUND"; + // Simulate the binding-missing failure as the better-sqlite3 default + // constructor throwing — this matches reality (`new Database()` throws + // "Could not locate the bindings file" when the prebuilt .node is absent) + // and reaches the adapter's `catch (nativeErr)`. A factory that itself + // throws is reported by vitest as a mock-setup error and never reaches + // the code under test. + const ThrowingDatabase = vi.fn(function ThrowingDatabase() { + throw bindingErr; + }); + vi.doMock("better-sqlite3", () => ({ + default: ThrowingDatabase, + })); + + // node:sqlite's DatabaseSync does not expose a boolean `open` property, + // so the mock intentionally omits it — the adapter tracks open state in + // a local closure and exposes it via a getter. + const mockNodeDb = { + prepare: vi.fn(() => createStatementMock()), + exec: vi.fn(), + close: vi.fn(), + }; + const DatabaseSync = vi.fn(function DatabaseSync() { + return mockNodeDb; + }); + vi.doMock("node:sqlite", () => ({ DatabaseSync })); + + const audit = await import("../audit.ts"); + + await audit.logToolCall("omniroute_get_health", { ok: true }, { ok: true }, 4, true); + expect(DatabaseSync).toHaveBeenCalledWith(dbFile); + expect(mockNodeDb.prepare).toHaveBeenCalled(); + + expect(audit.closeAuditDb()).toBe(true); + expect(mockNodeDb.exec).toHaveBeenCalledWith("PRAGMA wal_checkpoint(TRUNCATE)"); + expect(mockNodeDb.close).toHaveBeenCalledTimes(1); + + // Cache is cleared after close, so a second close is a no-op. + expect(audit.closeAuditDb()).toBe(false); + expect(mockNodeDb.close).toHaveBeenCalledTimes(1); + }); }); diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index f937df26c24..7a2db82b6da 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -88,6 +88,9 @@ describe("GLM Coding provider registry surfaces", () => { expect(PROVIDER_ID_TO_ALIAS.glm).toBe("glm"); expect(byProviderId).toEqual(byAlias); expect(byProviderId.map((model) => model.id)).toEqual([ + "glm-5.2", + "glm-5.2-high", + "glm-5.2-max", "glm-5.1", "glm-5", "glm-5-turbo", @@ -101,6 +104,30 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { + const models = getModelsByProviderId("glm"); + const get = (id: string) => models.find((m) => m.id === id); + + // Base model + const base = get("glm-5.2"); + expect(base).toBeDefined(); + expect(base?.contextLength).toBe(1000000); + expect(base?.maxOutputTokens).toBe(131072); + expect(base?.supportsReasoning).toBe(true); + expect(base?.toolCalling).toBe(true); + + // Effort tier aliases share the same specs + const high = get("glm-5.2-high"); + expect(high).toBeDefined(); + expect(high?.contextLength).toBe(1000000); + expect(high?.maxOutputTokens).toBe(131072); + + const max = get("glm-5.2-max"); + expect(max).toBeDefined(); + expect(max?.contextLength).toBe(1000000); + expect(max?.maxOutputTokens).toBe(131072); + }); + it("applies doc-backed context window overrides for GLM models", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); @@ -126,6 +153,9 @@ describe("GLM Coding provider registry surfaces", () => { expect(supportsToolCalling("glm/glm-5")).toBe(true); expect(supportsToolCalling("glm/glm-4.7-flash")).toBe(true); expect(supportsToolCalling("glm/glm-4.5-air")).toBe(true); + expect(supportsToolCalling("glm/glm-5.2")).toBe(true); + expect(supportsToolCalling("glm/glm-5.2-high")).toBe(true); + expect(supportsToolCalling("glm/glm-5.2-max")).toBe(true); expect(getPricingForModel("glm", "glm-5")).toEqual({ input: 1.0, @@ -148,6 +178,20 @@ describe("GLM Coding provider registry surfaces", () => { reasoning: 1.1, cache_creation: 0.2, }); + expect(getPricingForModel("glm", "glm-5.2")).toEqual({ + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }); + expect(getPricingForModel("glm", "glm-5.2-max")).toEqual({ + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }); }); it("keeps the repo-derived GLM inventory internally aligned across registry and pricing surfaces", () => { diff --git a/open-sse/mcp-server/audit.ts b/open-sse/mcp-server/audit.ts index 18897035e06..259ded1c49f 100644 --- a/open-sse/mcp-server/audit.ts +++ b/open-sse/mcp-server/audit.ts @@ -7,6 +7,7 @@ */ import { hashInput, summarizeOutput } from "./schemas/audit.ts"; +import { isNativeSqliteLoadError } from "../../src/lib/db/core.ts"; // ============ Database Connection ============ @@ -21,6 +22,61 @@ interface AuditDatabase { pragma: (sql: string) => unknown; close: () => void; open?: boolean; + driver?: "better-sqlite3" | "node:sqlite"; +} + +interface NodeSqliteDatabase { + prepare: (sql: string) => { + run: (...params: unknown[]) => { changes: number | bigint; lastInsertRowid: number | bigint }; + get: (...params: unknown[]) => unknown; + all: (...params: unknown[]) => unknown[]; + }; + exec: (sql: string) => void; + close: () => void; +} + +/** + * node:sqlite's `DatabaseSync` does NOT expose a boolean `open` property — + * `open` and `close` are methods on the prototype, and the only state + * surface is the `isOpen` getter. Track open state locally in a closure + * so the adapter's `AuditDatabase` contract (`open?: boolean`) is honored + * and `getCachedAuditDb()`'s truthy check doesn't return a closed handle + * after `closeAuditDb()`. + */ +function createNodeSqliteAuditAdapter(db: NodeSqliteDatabase): AuditDatabase { + let _isOpen = true; + return { + driver: "node:sqlite", + get open() { + return _isOpen; + }, + prepare(sql: string) { + const stmt = db.prepare(sql); + return { + get: (...params: unknown[]) => stmt.get(...params) as TRow | undefined, + all: (...params: unknown[]) => stmt.all(...params) as TRow[], + run: (...params: unknown[]) => stmt.run(...params), + }; + }, + pragma(pragmaSql: string) { + // node:sqlite has no .pragma() helper — route through .exec() for + // statement-shaped PRAGMAs (e.g. "wal_checkpoint(TRUNCATE)"). + try { + db.exec(`PRAGMA ${pragmaSql}`); + return null; + } catch (err) { + return err instanceof Error ? err.message : String(err); + } + }, + close: () => { + if (!_isOpen) return; + try { + db.close(); + } finally { + _isOpen = false; + } + }, + }; } declare global { @@ -153,6 +209,15 @@ function toString(value: unknown): string { /** * Lazy-load the database connection. * Uses the same SQLite database as the main OmniRoute app. + * + * Driver priority: + * 1. better-sqlite3 — fast native binding (when its compiled `.node` + * binary is present, see scripts/build/postinstall.mjs). + * 2. node:sqlite — built-in to Node 22.5+. Used as a transparent + * fallback so the MCP audit logger still works on installs where + * the better-sqlite3 binary failed to resolve (e.g. missing + * `dist/node_modules/better-sqlite3/build/Release/better_sqlite3.node` + * in some global-install / Docker scenarios). */ async function getDb(): Promise { const cachedDb = getCachedAuditDb(); @@ -173,12 +238,59 @@ async function getDb(): Promise { return null; } - const Database = (await import("better-sqlite3")).default as unknown as new ( - dbPath: string - ) => AuditDatabase; - const database = new Database(dbPath); - setCachedAuditDb(database); - return database; + // Try better-sqlite3 first (matches the main app's default driver). + try { + const Database = (await import("better-sqlite3")).default as unknown as new ( + dbPath: string + ) => AuditDatabase; + const database = new Database(dbPath); + setCachedAuditDb(database); + return database; + } catch (nativeErr) { + // Declared once at the top of the catch: nativeMessage is read both on + // the non-fallback bail-out and in the node:sqlite fallback warning + // further down. A block-scoped const inside the `if` below would be out + // of scope in the fallback path. + const nativeMessage = nativeErr instanceof Error ? nativeErr.message : String(nativeErr); + // Reuse the canonical detection helper from the main app's DB layer + // so we cover every ABI/binding failure mode the rest of the codebase + // already knows about: missing MODULE_NOT_FOUND, ERR_DLOPEN_FAILED, + // "Module did not self-register", "Cannot find module 'better-sqlite3'", + // the standard V8 "was compiled against a different Node.js version" + // message, and the bindings-loader "Could not locate the bindings file". + // Real errors (corrupt db, permission denied) still surface to the operator. + if (!isNativeSqliteLoadError(nativeErr)) { + console.error("[MCP Audit] Failed to connect to database:", nativeMessage); + return null; + } + // Fall back to Node's built-in sqlite (Node 22.5+). + const [maj, min] = (process.versions.node ?? "0.0").split(".").map(Number); + if (maj < 22 || (maj === 22 && (min ?? 0) < 5)) { + console.error( + `[MCP Audit] better-sqlite3 native binding unavailable and Node ${process.version} ` + + "has no built-in sqlite. Audit logging disabled. Fix: run " + + "`npm rebuild better-sqlite3` in the omniroute install root." + ); + return null; + } + try { + const { DatabaseSync } = (await import("node:sqlite")) as { + DatabaseSync: new (p: string) => NodeSqliteDatabase; + }; + const nodeDb = new DatabaseSync(dbPath); + const adapter = createNodeSqliteAuditAdapter(nodeDb); + setCachedAuditDb(adapter); + console.warn( + `[MCP Audit] better-sqlite3 binding unavailable — fell back to node:sqlite ` + + `(${nativeMessage.split("\n")[0]})` + ); + return adapter; + } catch (nodeErr) { + const nodeMessage = nodeErr instanceof Error ? nodeErr.message : String(nodeErr); + console.error("[MCP Audit] Failed to connect to database:", nodeMessage); + return null; + } + } } catch (err: unknown) { const message = err instanceof Error ? err.message : String(err); console.error("[MCP Audit] Failed to connect to database:", message); diff --git a/open-sse/package.json b/open-sse/package.json index 10ad729f0da..22fc9005dee 100644 --- a/open-sse/package.json +++ b/open-sse/package.json @@ -1,6 +1,6 @@ { "name": "@omniroute/open-sse", - "version": "3.8.25", + "version": "3.8.26", "description": "Express SSE sidecar for OmniRoute — handles streaming, protocol translation, and provider orchestration", "type": "module", "main": "index.js", diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 402f0330967..7d5be3aca87 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -828,6 +828,7 @@ const MAX_RR_COUNTERS = 500; const MAX_RESET_AWARE_CACHE = 200; const rrCounters = new Map(); +const rrStickyTargets = new Map(); const resetAwareConnectionCache = new Map< string, @@ -849,6 +850,50 @@ function normalizeModelEntry(entry: unknown): { model: string; weight: number } }; } +function clampStickyRoundRobinTargetLimit(value: unknown): number { + const numericValue = Number(value); + if (!Number.isFinite(numericValue)) return 1; + return Math.min(Math.max(Math.floor(numericValue), 1), 1000); +} + +function getStickyRoundRobinStartIndex( + comboName: string, + targets: ResolvedComboTarget[], + stickyLimit: number +): { startIndex: number; counter: number } { + const sticky = rrStickyTargets.get(comboName); + const stickyIndex = sticky + ? targets.findIndex((target) => target.executionKey === sticky.executionKey) + : -1; + if (stickyLimit > 1 && sticky && stickyIndex >= 0 && sticky.successCount < stickyLimit) { + return { startIndex: stickyIndex, counter: rrCounters.get(comboName) || 0 }; + } + + const counter = rrCounters.get(comboName) || 0; + return { startIndex: counter % targets.length, counter }; +} + +function recordStickyRoundRobinSuccess( + comboName: string, + target: ResolvedComboTarget, + stickyLimit: number, + targets: ResolvedComboTarget[] +): void { + const sticky = rrStickyTargets.get(comboName); + const successCount = sticky?.executionKey === target.executionKey ? sticky.successCount + 1 : 1; + if (successCount >= stickyLimit) { + const servedIndex = targets.findIndex((entry) => entry.executionKey === target.executionKey); + rrCounters.set( + comboName, + servedIndex >= 0 ? servedIndex + 1 : (rrCounters.get(comboName) || 0) + 1 + ); + rrStickyTargets.delete(comboName); + return; + } + + rrStickyTargets.set(comboName, { executionKey: target.executionKey, successCount }); +} + function getTargetProvider(modelStr: string, providerId?: string | null): string { const parsed = parseModel(modelStr); return providerId || parsed.provider || parsed.providerAlias || "unknown"; @@ -2994,7 +3039,8 @@ export async function expandAutoComboCandidatePool( (combo?.config as Record | undefined) || {}; - if (Array.isArray(localAutoConfig?.candidatePool)) return eligibleTargets; + if (Array.isArray(localAutoConfig?.candidatePool) && localAutoConfig.candidatePool.length > 0) + return eligibleTargets; try { const allConnections = await getProviderConnections({ isActive: true }); @@ -4700,14 +4746,38 @@ async function handleRoundRobinCombo({ log ); - // Get and increment atomic counter - const counter = rrCounters.get(combo.name) || 0; - if (!rrCounters.has(combo.name) && rrCounters.size >= MAX_RR_COUNTERS) { + // Sticky batch size at the combo level. Reuses the global `stickyRoundRobinLimit` + // setting so a single knob controls sticky batching for both account fallback and + // combo targets. Values <= 1 preserve the historical one-request-per-target rotation. + const stickyLimit = clampStickyRoundRobinTargetLimit( + (settings as Record | null)?.stickyRoundRobinLimit + ); + const stickyRoundRobinEnabled = stickyLimit > 1; + if ( + !rrCounters.has(combo.name) && + !rrStickyTargets.has(combo.name) && + rrCounters.size >= MAX_RR_COUNTERS + ) { const oldest = rrCounters.keys().next().value; - if (oldest !== undefined) rrCounters.delete(oldest); + if (oldest !== undefined) { + rrCounters.delete(oldest); + rrStickyTargets.delete(oldest); + } + } + // Ensure rrCounters has an entry for this combo so the eviction logic above + // applies to both maps even when sticky round-robin is enabled (in which + // case rrCounters isn't incremented per request). + if (!rrCounters.has(combo.name)) { + rrCounters.set(combo.name, 0); + } + const { startIndex, counter } = getStickyRoundRobinStartIndex( + combo.name, + filteredTargets, + stickyLimit + ); + if (!stickyRoundRobinEnabled) { + rrCounters.set(combo.name, counter + 1); } - rrCounters.set(combo.name, counter + 1); - const startIndex = counter % modelCount; const clientRequestedStream = body?.stream === true; const startTime = Date.now(); @@ -4903,6 +4973,10 @@ async function handleRoundRobinCombo({ recordProviderSuccess(provider, target.connectionId ?? undefined); } + if (stickyRoundRobinEnabled) { + recordStickyRoundRobinSuccess(combo.name, target, stickyLimit, filteredTargets); + } + if (provider) { const connId = target.connectionId || undefined; void (async () => { diff --git a/open-sse/utils/stream/claudeLifecycle.ts b/open-sse/utils/stream/claudeLifecycle.ts new file mode 100644 index 00000000000..3c2b34bb992 --- /dev/null +++ b/open-sse/utils/stream/claudeLifecycle.ts @@ -0,0 +1,180 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { JsonRecord } from "./types.ts"; + +export type ClaudeEmptyResponseLifecycle = { + hasMessageStart: boolean; + hasContentBlock: boolean; + hasMessageDelta: boolean; + hasMessageStop: boolean; + hasError: boolean; + syntheticContentInjected: boolean; + warningLogged: boolean; +}; + +export const SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT = ""; + +export function createClaudeEmptyResponseLifecycle(): ClaudeEmptyResponseLifecycle { + return { + hasMessageStart: false, + hasContentBlock: false, + hasMessageDelta: false, + hasMessageStop: false, + hasError: false, + syntheticContentInjected: false, + warningLogged: false, + }; +} + +export function getClaudeEventType(payload: unknown): string | null { + if (!payload || typeof payload !== "object") return null; + const type = (payload as JsonRecord).type; + return typeof type === "string" ? type : null; +} + +export function isClaudeEventPayload(payload: unknown): payload is JsonRecord { + return getClaudeEventType(payload) !== null; +} + +export function updateClaudeEmptyResponseLifecycle( + lifecycle: ClaudeEmptyResponseLifecycle, + payload: unknown +) { + const type = getClaudeEventType(payload); + if (!type) return; + + switch (type) { + case "message_start": + lifecycle.hasMessageStart = true; + break; + case "content_block_start": + case "content_block_delta": + case "content_block_stop": + lifecycle.hasContentBlock = true; + break; + case "message_delta": + lifecycle.hasMessageDelta = true; + break; + case "message_stop": + lifecycle.hasMessageStop = true; + break; + case "error": + lifecycle.hasError = true; + break; + default: + break; + } +} + +export function hasClaudeAssistantLifecycle(lifecycle: ClaudeEmptyResponseLifecycle): boolean { + return lifecycle.hasMessageStart || lifecycle.hasMessageDelta || lifecycle.hasMessageStop; +} + +export function shouldInjectClaudeEmptyResponseBeforeCurrentEvent( + lifecycle: ClaudeEmptyResponseLifecycle, + payload: unknown +): boolean { + const type = getClaudeEventType(payload); + if (!type || lifecycle.hasError || lifecycle.hasContentBlock) return false; + if (!hasClaudeAssistantLifecycle(lifecycle)) return false; + return type === "message_delta" || type === "message_stop"; +} + +export function shouldInjectClaudeEmptyResponseOnFlush(lifecycle: ClaudeEmptyResponseLifecycle): boolean { + if (lifecycle.hasError || lifecycle.hasContentBlock) return false; + return hasClaudeAssistantLifecycle(lifecycle); +} + +export function shouldInjectClaudeMissingFinalizersOnFlush( + lifecycle: ClaudeEmptyResponseLifecycle +): boolean { + if (lifecycle.hasError || !lifecycle.syntheticContentInjected) return false; + return !lifecycle.hasMessageDelta || !lifecycle.hasMessageStop; +} + +export function buildSyntheticClaudeEmptyResponseEvents( + lifecycle: ClaudeEmptyResponseLifecycle, + model: string | null, + options: { + includeContentBlock?: boolean; + includeMessageDelta?: boolean; + includeMessageStop?: boolean; + } = {} +): JsonRecord[] { + const { + includeContentBlock = true, + includeMessageDelta = false, + includeMessageStop = false, + } = options; + const events: JsonRecord[] = []; + const resolvedModel = typeof model === "string" && model ? model : "unknown"; + + if (includeContentBlock) { + if (!lifecycle.hasMessageStart) { + events.push({ + type: "message_start", + message: { + id: `msg_synthetic_${Date.now()}`, + type: "message", + role: "assistant", + model: resolvedModel, + content: [], + stop_reason: null, + stop_sequence: null, + usage: { input_tokens: 0, output_tokens: 0 }, + }, + }); + } + + events.push( + { + type: "content_block_start", + index: 0, + content_block: { type: "text", text: "" }, + }, + { + type: "content_block_delta", + index: 0, + delta: { + type: "text_delta", + text: SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT, + }, + }, + { + type: "content_block_stop", + index: 0, + } + ); + } + + if (includeMessageDelta) { + events.push({ + type: "message_delta", + delta: { stop_reason: "end_turn", stop_sequence: null }, + usage: { input_tokens: 0, output_tokens: 0 }, + }); + } + + if (includeMessageStop) { + events.push({ type: "message_stop" }); + } + + return events; +} + +export function restoreClaudePassthroughToolUseName(parsed: JsonRecord, toolNameMap: unknown): boolean { + if (!(toolNameMap instanceof Map)) return false; + if (!parsed || typeof parsed !== "object") return false; + + const block = + parsed.content_block && typeof parsed.content_block === "object" + ? (parsed.content_block as JsonRecord) + : null; + if (!block || block.type !== "tool_use" || typeof block.name !== "string") return false; + + const restoredName = toolNameMap.get(block.name) ?? block.name; + if (restoredName === block.name) return false; + block.name = restoredName; + return true; +} \ No newline at end of file diff --git a/open-sse/utils/stream/errors.ts b/open-sse/utils/stream/errors.ts new file mode 100644 index 00000000000..342cc5d8374 --- /dev/null +++ b/open-sse/utils/stream/errors.ts @@ -0,0 +1,87 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { asRecord } from "./utils.ts"; +import { JsonRecord, StreamFailurePayload } from "./types.ts"; + +export function toStreamFailureStatus(value: unknown): number | null { + if (typeof value === "number" && Number.isInteger(value) && value >= 400 && value <= 599) { + return value; + } + if (typeof value === "string" && /^\d{3}$/.test(value.trim())) { + const parsed = Number(value.trim()); + return parsed >= 400 && parsed <= 599 ? parsed : null; + } + return null; +} + +export function looksLikeStreamRateLimit(code: string, type: string, message: string): boolean { + const haystack = `${code} ${type} ${message}`.toLowerCase(); + return ( + haystack.includes("usage_limit_reached") || + haystack.includes("rate_limit") || + haystack.includes("rate limit") || + haystack.includes("quota") || + haystack.includes("too many requests") || + haystack.includes("limit reached") || + haystack.includes("limit has been reached") + ); +} + +function resolveErrorSource(response: JsonRecord, record: JsonRecord): JsonRecord { + const responseError = asRecord(response.error); + if (Object.keys(responseError).length) return responseError; + const recordError = asRecord(record.error); + if (Object.keys(recordError).length) return recordError; + return record; +} + +function resolveStreamFailureMessage(error: JsonRecord, record: JsonRecord): string { + if (typeof error.message === "string" && error.message.trim()) { + return error.message; + } + if (typeof record.message === "string" && record.message.trim()) { + return record.message; + } + return "Upstream failure"; +} + +function resolveStreamFailureStatus( + error: JsonRecord, + response: JsonRecord, + record: JsonRecord, + code: string, + type: string | undefined, + message: string +): number { + const candidates: unknown[] = [ + error.status_code, + error.status, + response.status_code, + response.status, + record.status_code, + record.status, + ]; + for (const candidate of candidates) { + const result = toStreamFailureStatus(candidate); + if (result !== null) return result; + } + return looksLikeStreamRateLimit(code, type || "", message) ? 429 : 502; +} + +export function normalizeStreamFailurePayload(payload: unknown): StreamFailurePayload | null { + const record = payload && typeof payload === "object" ? (payload as JsonRecord) : {}; + const response = asRecord(record.response); + const error = resolveErrorSource(response, record); + const code = typeof error.code === "string" ? error.code : "upstream_error"; + const type = typeof error.type === "string" ? error.type : undefined; + const message = resolveStreamFailureMessage(error, record); + const status = resolveStreamFailureStatus(error, response, record, code, type, message); + + return { + status, + message, + code, + ...(type ? { type } : {}), + }; +} diff --git a/open-sse/utils/stream/index.ts b/open-sse/utils/stream/index.ts new file mode 100644 index 00000000000..f22182abe95 --- /dev/null +++ b/open-sse/utils/stream/index.ts @@ -0,0 +1,9 @@ +export * from "./types.ts"; +export * from "./utils.ts"; +export * from "./responsesLifecycle.ts"; +export * from "./textualToolCalls.ts"; +export * from "./sseFormatters.ts"; +export * from "./errors.ts"; +export * from "./claudeLifecycle.ts"; +export * from "./openaiChunks.ts"; +export * from "./streamCore.ts"; diff --git a/open-sse/utils/stream/openaiChunks.ts b/open-sse/utils/stream/openaiChunks.ts new file mode 100644 index 00000000000..4c0b3a9c426 --- /dev/null +++ b/open-sse/utils/stream/openaiChunks.ts @@ -0,0 +1,10 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { JsonRecord } from "./types.ts"; + +export function getOpenAIIntermediateChunks(value: unknown): unknown[] { + if (!value || typeof value !== "object") return []; + const candidate = (value as JsonRecord)._openaiIntermediate; + return Array.isArray(candidate) ? candidate : []; +} \ No newline at end of file diff --git a/open-sse/utils/stream/responsesLifecycle.ts b/open-sse/utils/stream/responsesLifecycle.ts new file mode 100644 index 00000000000..1a37401a952 --- /dev/null +++ b/open-sse/utils/stream/responsesLifecycle.ts @@ -0,0 +1,203 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { stringifyIdValue } from "./utils.ts"; +import { JsonRecord } from "./types.ts"; + +function isNonArrayRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +export function normalizeResponsesOutputItemIds(item: unknown): unknown { + if (!item || typeof item !== "object" || Array.isArray(item)) { + return item; + } + + const record = item as JsonRecord; + let changed = false; + const normalized = { ...record }; + + const id = stringifyIdValue(record.id); + if (id !== null && record.id !== id) { + normalized.id = id; + changed = true; + } + + const callId = stringifyIdValue(record.call_id); + if (callId !== null && record.call_id !== callId) { + normalized.call_id = callId; + changed = true; + } + + return changed ? normalized : item; +} + +export function normalizeResponsesSseIds(payload: JsonRecord): boolean { + let changed = false; + + for (const key of ["response_id", "item_id", "call_id"] as const) { + const value = stringifyIdValue(payload[key]); + if (value !== null && payload[key] !== value) { + payload[key] = value; + changed = true; + } + } + + if (isNonArrayRecord(payload.item)) { + const normalizedItem = normalizeResponsesOutputItemIds(payload.item); + if (normalizedItem !== payload.item) { + payload.item = normalizedItem; + changed = true; + } + } + + if (isNonArrayRecord(payload.response)) { + const response = payload.response as JsonRecord; + let responseChanged = false; + const normalizedResponse = { ...response }; + + const responseId = stringifyIdValue(response.id); + if (responseId !== null && response.id !== responseId) { + normalizedResponse.id = responseId; + responseChanged = true; + } + + if (Array.isArray(response.output)) { + const normalizedOutput = response.output.map(normalizeResponsesOutputItemIds); + if (normalizedOutput.some((item, index) => item !== response.output[index])) { + normalizedResponse.output = normalizedOutput; + responseChanged = true; + } + } + + if (responseChanged) { + payload.response = normalizedResponse; + changed = true; + } + } + + return changed; +} + +export const PENDING_REQUEST_CLEARED_MARKER = "__omniroutePendingRequestCleared"; + +export function markPendingRequestCleared(error: Error): Error { + (error as Error & Record)[PENDING_REQUEST_CLEARED_MARKER] = true; + return error; +} + +export function buildResponsesOutputItemKey(item: unknown): string | null { + if (!item || typeof item !== "object" || Array.isArray(item)) { + return null; + } + + const record = item as JsonRecord; + const type = typeof record.type === "string" ? record.type : ""; + const id = stringifyIdValue(record.id) ?? ""; + const callId = stringifyIdValue(record.call_id) ?? ""; + const outputIndex = typeof record.output_index === "number" ? record.output_index : ""; + const name = typeof record.name === "string" ? record.name : ""; + + if (!type && !id && !callId) { + return null; + } + + return `${type}:${id}:${callId}:${outputIndex}:${name}`; +} + +export function pushUniqueResponsesOutputItems(target: unknown[], items: readonly unknown[]) { + const seen = new Set(); + + for (const existingItem of target) { + const key = buildResponsesOutputItemKey(existingItem); + if (key) { + seen.add(key); + } + } + + for (const item of items) { + const key = buildResponsesOutputItemKey(item); + if (key && seen.has(key)) { + continue; + } + + target.push(item); + if (key) { + seen.add(key); + } + } +} + +/** + * Lifecycle event types in OpenAI Responses API streams whose `response` + * payload is a snapshot of the request (echoes back `instructions` + `tools`). + */ +export const RESPONSES_LIFECYCLE_EVENT_TYPES = new Set([ + "response.created", + "response.in_progress", + "response.completed", +]); + +/** + * Backfill `parsed.response.output` on a `response.completed` event from the + * snapshots accumulated as the stream progressed (`response.output_item.done`). + * + * Why: when the upstream request runs with `store: false`, OpenAI's Responses + * API leaves `response.output` empty in the final `response.completed` + * snapshot — clients that rebuild assistant messages from that snapshot + * (notably the GitHub Copilot CLI 1.0.36) end up with `choices: []` and never + * trigger tool execution. Codex CLI and others that consume per-item events + * are unaffected; backfilling the array makes both styles work. + * + * Returns true when `parsed.response.output` was empty and got replaced, so + * the caller can re-serialize. + */ +export function backfillResponsesCompletedOutput( + parsed: unknown, + collectedItems: readonly unknown[] +): boolean { + if (!collectedItems.length) return false; + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false; + const obj = parsed as Record; + if (obj.type !== "response.completed") return false; + const resp = obj.response; + if (!resp || typeof resp !== "object" || Array.isArray(resp)) return false; + const r = resp as Record; + const existing = r.output; + if (Array.isArray(existing) && existing.length > 0) return false; + r.output = collectedItems.slice(); + return true; +} + +/** + * Strip the request echo (`instructions`, `tools`) from `parsed.response` + * on Responses API lifecycle events. + * + * Why: those fields can balloon the SSE message past 100 KB when the request + * carries large tool definitions / instructions. Some clients (notably the + * GitHub Copilot CLI) cannot process oversized SSE events and stop rendering + * mid-stream. The fields are pure echo of the original request — clients + * already hold the original locally — so removing them is observably safe. + * + * Returns true when the payload was modified and must be re-serialized. + */ +export function stripResponsesLifecycleEcho(parsed: unknown): boolean { + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false; + const obj = parsed as Record; + if (typeof obj.type !== "string" || !RESPONSES_LIFECYCLE_EVENT_TYPES.has(obj.type)) { + return false; + } + const resp = obj.response; + if (!resp || typeof resp !== "object" || Array.isArray(resp)) return false; + const r = resp as Record; + let changed = false; + if ("instructions" in r) { + delete r.instructions; + changed = true; + } + if ("tools" in r) { + delete r.tools; + changed = true; + } + return changed; +} diff --git a/open-sse/utils/stream/sseFormatters.ts b/open-sse/utils/stream/sseFormatters.ts new file mode 100644 index 00000000000..0d2c759118b --- /dev/null +++ b/open-sse/utils/stream/sseFormatters.ts @@ -0,0 +1,90 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { asRecord } from "./utils.ts"; +import { JsonRecord, ToolCall } from "./types.ts"; + +/* @testonly */ export function toStreamingToolCallDelta(toolCall: ToolCall) { + return { + index: toolCall.index, + id: toolCall.id != null ? String(toolCall.id) : null, + type: toolCall.type, + function: { + name: toolCall.function.name, + arguments: toolCall.function.arguments, + }, + }; +} + +/* @testonly */ export function toResponsesFunctionCallItem(toolCall: ToolCall) { + return { + type: "function_call", + id: (toolCall.id != null ? String(toolCall.id) : null) || `fc_${toolCall.index}`, + call_id: (toolCall.id != null ? String(toolCall.id) : null) || `call_${toolCall.index}`, + name: toolCall.function.name, + arguments: toolCall.function.arguments, + status: "completed", + }; +} + +export function buildResponsesFunctionCallEvents(toolCall: ToolCall) { + const item = toResponsesFunctionCallItem(toolCall); + return [ + { + type: "response.output_item.added", + output_index: toolCall.index, + item, + }, + { + type: "response.function_call_arguments.done", + item_id: item.id, + output_index: toolCall.index, + arguments: toolCall.function.arguments, + }, + { + type: "response.output_item.done", + output_index: toolCall.index, + item, + }, + ]; +} + +export function formatSSEDataEvents(events: unknown[]) { + return events.map((event) => `data: ${JSON.stringify(event)}\n`).join("\n"); +} + +export function toChatCompletionChunkWithToolCall(base: JsonRecord, toolCall: ToolCall) { + const choice = asRecord(Array.isArray(base.choices) ? base.choices[0] : null); + const delta = { ...asRecord(choice.delta) }; + delete delta.content; + delete delta.reasoning_content; + return { + ...base, + choices: [ + { + ...choice, + index: typeof choice.index === "number" ? choice.index : 0, + delta: { + ...delta, + tool_calls: [toStreamingToolCallDelta(toolCall)], + }, + finish_reason: null, + }, + ], + }; +} + +export function toResponsesCompletedWithToolCalls(parsed: JsonRecord, toolCalls: ToolCall[]) { + const response = asRecord(parsed.response); + const existingOutput = Array.isArray(response.output) ? response.output : []; + return { + ...parsed, + response: { + ...response, + output: [ + ...existingOutput, + ...toolCalls.map((toolCall) => toResponsesFunctionCallItem(toolCall)), + ], + }, + }; +} \ No newline at end of file diff --git a/open-sse/utils/stream/streamCore.ts b/open-sse/utils/stream/streamCore.ts new file mode 100644 index 00000000000..2c736ab45d0 --- /dev/null +++ b/open-sse/utils/stream/streamCore.ts @@ -0,0 +1,2215 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { translateResponse, initState } from "../../translator/index.ts"; +import { v4 as uuidv4 } from "uuid"; +import { FORMATS } from "../../translator/formats.ts"; +import { generateSessionId } from "../../services/sessionManager.ts"; +import { trackPendingRequest, appendRequestLog } from "@/lib/usage/usageHistory.ts"; +import { calculateCost } from "@/lib/usage/costCalculator"; +import { buildOmniRouteSseMetadataComment } from "@/domain/omnirouteResponseMeta"; +import { createStructuredSSECollector } from "../streamPayloadCollector.ts"; +import { + STREAM_IDLE_TIMEOUT_MS, + FETCH_BODY_TIMEOUT_MS, + HTTP_STATUS, +} from "../../config/constants.ts"; +import { parseSSELine } from "../streamHelpers.ts"; +import { recordToolLatency } from "../../services/toolLatencyTracker.ts"; +import { processBufferedPassthroughLine } from "../passthroughTailProcessor.ts"; +import { normalizeStreamFailurePayload } from "./errors.ts"; +import { buildErrorBody } from "../error.ts"; +import { SSEStreamContext } from "./types.ts"; + +import { + parseTextualToolCallFromContent, + containsTextualToolCallCandidate, + containsMalformedTextualToolCall, + extractAllowedToolNames, + collectPassthroughTextualToolCall, +} from "./textualToolCalls.ts"; +import { getOpenAIIntermediateChunks } from "./openaiChunks.ts"; +import { + SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT, + createClaudeEmptyResponseLifecycle, + getClaudeEventType, + isClaudeEventPayload, + updateClaudeEmptyResponseLifecycle, + shouldInjectClaudeEmptyResponseBeforeCurrentEvent, + shouldInjectClaudeEmptyResponseOnFlush, + shouldInjectClaudeMissingFinalizersOnFlush, + buildSyntheticClaudeEmptyResponseEvents, + restoreClaudePassthroughToolUseName, +} from "./claudeLifecycle.ts"; +import { + normalizeResponsesSseIds, + markPendingRequestCleared, + pushUniqueResponsesOutputItems, + backfillResponsesCompletedOutput, + stripResponsesLifecycleEcho, +} from "./responsesLifecycle.ts"; +import { + buildResponsesFunctionCallEvents, + formatSSEDataEvents, + toChatCompletionChunkWithToolCall, + toResponsesCompletedWithToolCalls, +} from "./sseFormatters.ts"; +import { stringifyIdValue, asRecord, appendBoundedText, STREAM_MODE } from "./utils.ts"; +import { + JsonRecord, + StreamLogger, + StreamCompletePayload, + StreamFailurePayload, + StreamOptions, + TranslateState, + ToolCall, + UsageTokenRecord, +} from "./types.ts"; +import { normalizeStreamFailurePayload } from "./errors.ts"; +import { buildErrorBody } from "../error.ts"; + +// Module-level helpers extracted from createSSEStream closures to keep +// cyclomatic complexity of individual functions under the 15-node gate. + +function getResponsesReasoningSummaryTextModule(item: Record): string { + return Array.isArray(item.summary) + ? item.summary + .map((part) => { + if (!part || typeof part !== "object" || Array.isArray(part)) { + return ""; + } + return typeof (part as Record).text === "string" + ? ((part as Record).text as string) + : ""; + }) + .join("") + : ""; +} + +function getResponsesReasoningKeyModule( + payload: Record, + passthroughResponsesId: string | null +): string | null { + const itemId = stringifyIdValue(payload.item_id); + if (itemId) { + return itemId; + } + const item = + payload.item && typeof payload.item === "object" && !Array.isArray(payload.item) + ? (payload.item as Record) + : null; + const outputItemId = item ? stringifyIdValue(item.id) : null; + if (outputItemId) { + return outputItemId; + } + const responseId = stringifyIdValue(payload.response_id) || passthroughResponsesId; + const outputIndex = + typeof payload.output_index === "number" && Number.isInteger(payload.output_index) + ? payload.output_index + : null; + return responseId !== null && outputIndex !== null ? `${responseId}:${outputIndex}` : null; +} + +function ensureVisibleResponsesReasoningSummaryModule(payload: Record): boolean { + const item = + payload.item && typeof payload.item === "object" && !Array.isArray(payload.item) + ? (payload.item as Record) + : null; + if (!item || item.type !== "reasoning") { + return false; + } + if (getResponsesReasoningSummaryTextModule(item)) { + return false; + } + const hasEncryptedReasoning = + typeof item.encrypted_content === "string" && item.encrypted_content.length > 0; + if (!hasEncryptedReasoning) { + return false; + } + item.summary = [ + { + type: "summary_text", + text: "Codex is reasoning, but the upstream Responses API exposed this reasoning block only as encrypted state. OmniRoute cannot recover the private reasoning text.", + }, + ]; + return true; +} + +function computeReasoningIdsModule( + item: Record, + payload: Record, + reasoningKey: string +): { itemId: string; outputIndex: number } { + const itemId = typeof item.id === "string" && item.id ? item.id : reasoningKey; + const outputIndex = + typeof payload.output_index === "number" && Number.isInteger(payload.output_index) + ? payload.output_index + : 0; + return { itemId, outputIndex }; +} + +function maybeExtractOpenAIThinking( + itemSanitized: Record, + sourceFormat: string +): Record | null { + const isResponsesEvent = + typeof itemSanitized?.event === "string" && itemSanitized.event.startsWith("response."); + if (sourceFormat === FORMATS.OPENAI && !isResponsesEvent) { + const sanitized = sanitizeStreamingChunk(itemSanitized) as Record; + const delta = sanitized?.choices?.[0]?.delta; + if (delta?.content && typeof delta.content === "string") { + const { content, thinking } = extractThinkingFromContent(delta.content); + delta.content = content; + if (thinking && !delta.reasoning_content) { + delta.reasoning_content = thinking; + } + } + return sanitized; + } + return null; +} + +function maybeFillFinishChunkUsage( + itemSanitized: Record, + state: TranslateState | null | undefined, + isFinishChunk: boolean, + body: Record | null | undefined, + totalContentLength: number, + sourceFormat: string +): void { + if (!state?.finishReason || !isFinishChunk) { + return; + } + if (!hasValidUsage(itemSanitized.usage) && totalContentLength > 0) { + const estimated = estimateUsage(body, totalContentLength, sourceFormat); + itemSanitized.usage = filterUsageForFormat(estimated, sourceFormat); + state.usage = estimated; + } else if (state.usage) { + const buffered = addBufferToUsage(state.usage); + itemSanitized.usage = filterUsageForFormat(buffered, sourceFormat); + } +} + +function maybeEmitClaudeLifecycleEvents( + itemSanitized: Record, + claudeEmptyResponseLifecycle: Record, + sourceFormat: string +): { shouldInjectEmptyResponse: boolean; eventType: string | null } { + let shouldInjectEmptyResponse = false; + let eventType: string | null = null; + if ( + sourceFormat === FORMATS.CLAUDE && + shouldInjectClaudeEmptyResponseBeforeCurrentEvent( + claudeEmptyResponseLifecycle as Parameters< + typeof shouldInjectClaudeEmptyResponseBeforeCurrentEvent + >[0], + itemSanitized + ) + ) { + shouldInjectEmptyResponse = true; + eventType = getClaudeEventType(itemSanitized); + } + if (sourceFormat === FORMATS.CLAUDE && isClaudeEventPayload(itemSanitized)) { + updateClaudeEmptyResponseLifecycle( + claudeEmptyResponseLifecycle as Parameters[0], + itemSanitized + ); + } + return { shouldInjectEmptyResponse, eventType }; +} + +const CREATE_SSE_STREAM_DEFAULTS = { + mode: STREAM_MODE.TRANSLATE, + clientResponseFormat: null, + copilotCompatibleReasoning: false, + provider: null, + reqLogger: null, + toolNameMap: null, + model: null, + connectionId: null, + apiKeyInfo: null, + body: null, + onComplete: null, + onFailure: null, +}; + +function buildSSEStreamContext(options: StreamOptions): SSEStreamContext { + const { + mode, + targetFormat, + sourceFormat, + clientResponseFormat, + copilotCompatibleReasoning, + provider, + reqLogger, + toolNameMap, + model, + connectionId, + apiKeyInfo, + body, + onComplete, + onFailure, + } = { ...CREATE_SSE_STREAM_DEFAULTS, ...options }; + const signatureNamespace = connectionId; + + const clientExpectsResponsesStream = + (mode === STREAM_MODE.PASSTHROUGH + ? clientResponseFormat === FORMATS.OPENAI_RESPONSES + : sourceFormat === FORMATS.OPENAI_RESPONSES) === true; + + const clientExpectsClaudeStream = + (mode === STREAM_MODE.PASSTHROUGH + ? clientResponseFormat === FORMATS.CLAUDE + : sourceFormat === FORMATS.CLAUDE) === true; + + const shouldEmitDoneTerminator = !clientExpectsResponsesStream && !clientExpectsClaudeStream; + + let buffer = ""; + let usage: UsageTokenRecord | null = null; + let passthroughHasToolCalls = false; + const passthroughToolCalls = new Map(); + let passthroughToolCallSeq = 0; + const allowedToolNames = extractAllowedToolNames(body); + let skipPassthroughEvent = false; + + const state: TranslateState | null = + mode === STREAM_MODE.TRANSLATE + ? { + ...(initState(sourceFormat) as TranslateState), + provider, + toolNameMap, + signatureNamespace, + copilotCompatibleReasoning, + accumulatedContent: "", + } + : null; + + let totalContentLength = 0; + let passthroughAccumulatedContent = ""; + let passthroughAccumulatedReasoning = ""; + let passthroughBufferedTextualToolCallContent = ""; + const passthroughResponsesOutputItems: unknown[] = []; + const passthroughResponsesPendingFunctionCalls = new Map(); + let passthroughResponsesId: string | null = null; + let passthroughResponsesCurrentFunctionCallKey: string | null = null; + const passthroughResponsesReasoningSummarySeen = new Set(); + const streamStartedAt = Date.now(); + + let lastToolCallChunkTime: number | null = null; + let toolFinishTime: number | null = null; + let contentAfterToolSeen = false; + + const sessionId = generateSessionId(body as Parameters[0], { + provider: provider ?? undefined, + connectionId: connectionId ?? undefined, + }); + let pendingToolFinishTime: number | null = null; + try { + pendingToolFinishTime = consumeToolFinishTime(sessionId); + } catch {} + + let doneSent = false; + const providerPayloadCollector = createStructuredSSECollector({ + stage: "provider_response", + }); + const clientPayloadCollector = createStructuredSSECollector({ + stage: "client_response", + }); + const requestRecord = asRecord(body); + const requestStreamOptions = asRecord( + requestRecord.stream_options ?? requestRecord.streamOptions + ); + const expectsOpenAIUsageOnlyChunk = + requestStreamOptions.include_usage === true || requestStreamOptions.includeUsage === true; + + const decoder = new TextDecoder(); + const encoder = new TextEncoder(); + + let lastChunkTime = Date.now(); + let idleTimer: ReturnType | null = null; + let streamTimedOut = false; + const claudeEmptyResponseLifecycle = createClaudeEmptyResponseLifecycle() as Record< + string, + unknown + >; + let pendingPassthroughEventLine: string | null = null; + let pendingPassthroughEventEmitted = false; + + const clearIdleTimer = () => { + if (idleTimer) { + clearInterval(idleTimer); + idleTimer = null; + } + }; + + const clearPendingPassthroughEvent = () => { + pendingPassthroughEventLine = null; + pendingPassthroughEventEmitted = false; + }; + + const maybePrefixPendingPassthroughEvent = (output: string, line: string) => { + if (!pendingPassthroughEventLine || !line.startsWith("data:")) { + return output; + } + if (!pendingPassthroughEventEmitted) { + pendingPassthroughEventEmitted = true; + return `${pendingPassthroughEventLine}\n${output}`; + } + return output; + }; + + const applyTextualToolCallStreamingGuard = (parsed: Record) => { + const choice = Array.isArray((parsed as JsonRecord).choices) + ? (((parsed as JsonRecord).choices as unknown[])[0] as JsonRecord | undefined) + : undefined; + const delta = asRecord(choice?.delta); + let textualToolCallConverted = false; + + if (typeof delta?.content === "string") { + const incomingContent = delta.content; + const bufferedCandidate = passthroughBufferedTextualToolCallContent + incomingContent; + if ( + passthroughBufferedTextualToolCallContent || + containsTextualToolCallCandidate(incomingContent) + ) { + const parsedCandidate = parseTextualToolCallCandidate(bufferedCandidate); + if (parsedCandidate?.kind === "complete") { + const collectedToolCall = collectPassthroughTextualToolCall( + bufferedCandidate, + passthroughToolCalls, + allowedToolNames + ); + if (collectedToolCall) { + parsed = toChatCompletionChunkWithToolCall(parsed, collectedToolCall); + passthroughHasToolCalls = true; + } else { + delete delta.content; + delete delta.reasoning_content; + } + textualToolCallConverted = true; + passthroughBufferedTextualToolCallContent = ""; + } else if (parsedCandidate?.kind === "partial") { + passthroughBufferedTextualToolCallContent = appendBoundedText( + passthroughBufferedTextualToolCallContent, + incomingContent + ); + textualToolCallConverted = true; + delta.content = ""; + } else { + if (passthroughBufferedTextualToolCallContent) { + delta.content = passthroughBufferedTextualToolCallContent + incomingContent; + textualToolCallConverted = true; + } + passthroughAccumulatedContent = appendBoundedText( + passthroughAccumulatedContent, + passthroughBufferedTextualToolCallContent + incomingContent + ); + passthroughBufferedTextualToolCallContent = ""; + } + } else { + passthroughAccumulatedContent = appendBoundedText( + passthroughAccumulatedContent, + incomingContent + ); + } + } + + return { parsed, textualToolCallConverted }; + }; + + const emitSyntheticClaudeEmptyResponse = ( + controller: TransformStreamDefaultController, + options: { + includeContentBlock?: boolean; + includeMessageDelta?: boolean; + includeMessageStop?: boolean; + } = {} + ) => { + const events = buildSyntheticClaudeEmptyResponseEvents( + claudeEmptyResponseLifecycle, + model, + options + ); + if (events.length === 0) return; + + if (!claudeEmptyResponseLifecycle.warningLogged) { + claudeEmptyResponseLifecycle.warningLogged = true; + console.warn( + `[STREAM] Injecting synthetic Claude SSE response for empty upstream output (${provider || "provider"}:${model || "unknown"})` + ); + } + + if (options.includeContentBlock !== false) { + claudeEmptyResponseLifecycle.syntheticContentInjected = true; + if (!passthroughAccumulatedContent.trim()) { + passthroughAccumulatedContent = SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT; + } + if (state?.accumulatedContent !== undefined && !state.accumulatedContent.trim()) { + state.accumulatedContent = SYNTHETIC_CLAUDE_EMPTY_RESPONSE_TEXT; + } + } + + for (const event of events) { + updateClaudeEmptyResponseLifecycle(claudeEmptyResponseLifecycle, event); + clientPayloadCollector.push(event); + const output = formatSSE(event, FORMATS.CLAUDE); + reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(encoder.encode(output)); + } + }; + + const emitTranslatedClientItem = ( + controller: TransformStreamDefaultController, + item: Record + ) => { + let itemSanitized: Record = item; + + const sanitized = maybeExtractOpenAIThinking(itemSanitized, sourceFormat); + if (sanitized) { + itemSanitized = sanitized; + } + + if (!hasValuableContent(itemSanitized, sourceFormat)) { + return; + } + + const isFinishChunk = + itemSanitized.type === "message_delta" || itemSanitized.choices?.[0]?.finish_reason; + maybeFillFinishChunkUsage( + itemSanitized, + state, + isFinishChunk, + body, + totalContentLength, + sourceFormat + ); + + const { shouldInjectEmptyResponse, eventType } = maybeEmitClaudeLifecycleEvents( + itemSanitized, + claudeEmptyResponseLifecycle, + sourceFormat + ); + if (shouldInjectEmptyResponse) { + emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: true, + includeMessageDelta: + eventType === "message_stop" && + !(claudeEmptyResponseLifecycle as Record).hasMessageDelta, + includeMessageStop: false, + }); + } + + const output = formatSSE(itemSanitized, sourceFormat); + clientPayloadCollector.push(itemSanitized); + reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(encoder.encode(output)); + }; + + const emitFinalSseMetadata = async ( + controller: TransformStreamDefaultController, + finalUsage: UsageTokenRecord | Record | null | undefined + ) => { + const costUsd = finalUsage ? await calculateCost(provider, model, finalUsage) : 0; + const comment = buildOmniRouteSseMetadataComment({ + provider, + model, + cacheHit: false, + latencyMs: Date.now() - streamStartedAt, + usage: finalUsage, + costUsd, + }); + if (!comment) return; + reqLogger?.appendConvertedChunk?.(comment); + controller.enqueue(encoder.encode(comment)); + }; + + const getResponsesReasoningKey = (payload: Record): string | null => + getResponsesReasoningKeyModule(payload, passthroughResponsesId); + + const getResponsesReasoningSummaryText = getResponsesReasoningSummaryTextModule; + + const ensureVisibleResponsesReasoningSummary = ensureVisibleResponsesReasoningSummaryModule; + + const emitSyntheticResponsesReasoningSummary = ( + controller: TransformStreamDefaultController, + payload: Record + ) => { + const item = + payload.item && typeof payload.item === "object" && !Array.isArray(payload.item) + ? (payload.item as Record) + : null; + if (!item || item.type !== "reasoning") { + return; + } + + ensureVisibleResponsesReasoningSummary(payload); + const visibleSummary = getResponsesReasoningSummaryText(item); + + if (!visibleSummary) { + return; + } + + const reasoningKey = getResponsesReasoningKey(payload); + if (!reasoningKey || passthroughResponsesReasoningSummarySeen.has(reasoningKey)) { + return; + } + passthroughResponsesReasoningSummarySeen.add(reasoningKey); + + const { itemId, outputIndex } = computeReasoningIdsModule(item, payload, reasoningKey); + + const syntheticEvents = [ + { + event: "response.reasoning_summary_text.delta", + body: { + type: "response.reasoning_summary_text.delta", + item_id: itemId, + output_index: outputIndex, + summary_index: 0, + delta: visibleSummary, + }, + }, + { + event: "response.reasoning_summary_part.done", + body: { + type: "response.reasoning_summary_part.done", + item_id: itemId, + output_index: outputIndex, + summary_index: 0, + part: { type: "summary_text", text: visibleSummary }, + }, + }, + ]; + + for (const syntheticEvent of syntheticEvents) { + clientPayloadCollector.push(syntheticEvent.body); + const output = `event: ${syntheticEvent.event}\ndata: ${JSON.stringify(syntheticEvent.body)}\n\n`; + reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(encoder.encode(output)); + } + }; + + return { + mode: mode as string, + targetFormat: targetFormat as string | undefined, + sourceFormat: sourceFormat as string | undefined, + clientResponseFormat: clientResponseFormat as string | null, + copilotCompatibleReasoning: copilotCompatibleReasoning as boolean, + provider: provider as string | null, + reqLogger: reqLogger as StreamLogger | null, + toolNameMap, + model: model as string | null, + connectionId: connectionId as string | null, + apiKeyInfo, + body, + onComplete: onComplete as ((payload: StreamCompletePayload) => void) | null, + onFailure: onFailure as ((payload: StreamFailurePayload) => void | Promise) | null, + clientExpectsResponsesStream, + clientExpectsClaudeStream, + shouldEmitDoneTerminator, + expectsOpenAIUsageOnlyChunk, + signatureNamespace: signatureNamespace as string | null, + buffer, + usage, + passthroughHasToolCalls, + passthroughToolCalls, + passthroughToolCallSeq, + allowedToolNames, + skipPassthroughEvent, + state, + totalContentLength, + passthroughAccumulatedContent, + passthroughAccumulatedReasoning, + passthroughBufferedTextualToolCallContent, + passthroughResponsesOutputItems, + passthroughResponsesPendingFunctionCalls, + passthroughResponsesId, + passthroughResponsesCurrentFunctionCallKey, + passthroughResponsesReasoningSummarySeen, + streamStartedAt, + lastToolCallChunkTime, + toolFinishTime, + contentAfterToolSeen, + sessionId, + pendingToolFinishTime, + doneSent, + pendingPassthroughEventLine, + pendingPassthroughEventEmitted, + lastChunkTime, + streamTimedOut, + decoder, + encoder, + idleTimer, + claudeEmptyResponseLifecycle, + providerPayloadCollector, + clientPayloadCollector, + requestRecord, + requestStreamOptions, + clearIdleTimer, + clearPendingPassthroughEvent, + maybePrefixPendingPassthroughEvent, + applyTextualToolCallStreamingGuard, + emitSyntheticClaudeEmptyResponse, + emitTranslatedClientItem, + emitFinalSseMetadata, + getResponsesReasoningKey, + getResponsesReasoningSummaryText, + ensureVisibleResponsesReasoningSummary, + emitSyntheticResponsesReasoningSummary, + }; +} + +/** + * Create unified SSE transform stream with idle timeout protection. + * If the upstream provider stops sending data for STREAM_IDLE_TIMEOUT_MS, + * the stream emits an error event and closes to prevent indefinite hanging. + * + * @param {object} options + * @param {string} options.mode - Stream mode: translate, passthrough + * @param {string} options.targetFormat - Provider format (for translate mode) + * @param {string} options.sourceFormat - Client format (for translate mode) + * @param {string} options.provider - Provider name + * @param {object} options.reqLogger - Request logger instance + * @param {string} options.model - Model name + * @param {string} options.connectionId - Connection ID for usage tracking + * @param {object|null} options.apiKeyInfo - API key metadata for usage attribution + * @param {object} options.body - Request body (for input token estimation) + * @param {function} options.onComplete - Callback when stream finishes: ({ status, usage }) => void + */ +export function createSSEStream(options: StreamOptions = {}) { + const ctx = buildSSEStreamContext(options); + return new TransformStream( + { + start(controller) { + // Start idle watchdog — checks every 10s if ctx.provider has stopped sending + if (STREAM_IDLE_TIMEOUT_MS > 0) { + ctx.idleTimer = setInterval(() => { + if (!ctx.streamTimedOut && Date.now() - ctx.lastChunkTime > STREAM_IDLE_TIMEOUT_MS) { + ctx.streamTimedOut = true; + ctx.clearIdleTimer(); + const timeoutMsg = `[STREAM] Idle timeout: no data from ${ctx.provider || "provider"} for ${STREAM_IDLE_TIMEOUT_MS}ms (ctx.model: ${ctx.model || "unknown"})`; + console.warn(timeoutMsg); + trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false); + appendRequestLog({ + model: ctx.model, + provider: ctx.provider, + connectionId: ctx.connectionId, + status: `FAILED ${HTTP_STATUS.GATEWAY_TIMEOUT}`, + }).catch(() => {}); + const timeoutError = new Error(timeoutMsg); + timeoutError.name = "StreamIdleTimeoutError"; + controller.error(markPendingRequestCleared(timeoutError)); + } + }, 10_000); + } + }, + + transform(chunk, controller) { + if (ctx.streamTimedOut) return; + ctx.lastChunkTime = Date.now(); + const text = ctx.decoder.decode(chunk, { stream: true }); + ctx.buffer += text; + ctx.reqLogger?.appendProviderChunk?.(text); + + const lines = ctx.buffer.split("\n"); + ctx.buffer = lines.pop() || ""; + + for (const line of lines) { + const trimmed = line.trim(); + + // Passthrough ctx.mode: normalize and forward + if (ctx.mode === STREAM_MODE.PASSTHROUGH) { + let output: string; + let injectedUsage = false; + let clientPayload: unknown = null; + let failurePayload: StreamFailurePayload | null = null; + + if (ctx.skipPassthroughEvent) { + if (!trimmed) { + ctx.skipPassthroughEvent = false; + ctx.clearPendingPassthroughEvent(); + } + continue; + } + + // Drop whole keepalive event blocks — strict OpenAI-compatible SDKs + // try to JSON.parse empty keepalive payloads and crash. + if (/^event:\s*keepalive\b/i.test(trimmed)) { + ctx.skipPassthroughEvent = true; + ctx.clearPendingPassthroughEvent(); + continue; + } + + if (/^event:/i.test(trimmed)) { + if (ctx.pendingPassthroughEventLine && !ctx.pendingPassthroughEventEmitted) { + const pendingOutput = `${ctx.pendingPassthroughEventLine}\n`; + ctx.reqLogger?.appendConvertedChunk?.(pendingOutput); + controller.enqueue(ctx.encoder.encode(pendingOutput)); + } + + const eventType = trimmed.replace(/^event:\s*/i, ""); + if ( + shouldInjectClaudeEmptyResponseBeforeCurrentEvent( + ctx.claudeEmptyResponseLifecycle, + { + type: eventType, + } + ) + ) { + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: true, + includeMessageDelta: + eventType === "message_stop" && + !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: false, + }); + } + + ctx.pendingPassthroughEventLine = line; + ctx.pendingPassthroughEventEmitted = false; + continue; + } + + if (trimmed.startsWith("data:")) { + const providerPayload = parseSSELine(trimmed); + if (providerPayload) { + ctx.providerPayloadCollector.push(providerPayload); + if ((providerPayload as { done?: unknown }).done === true) { + continue; + } + } + } + + if (trimmed.startsWith("data:") && trimmed.slice(5).trim() === "[DONE]") { + continue; + } + + if (trimmed.startsWith("data:") && trimmed.slice(5).trim() !== "[DONE]") { + try { + let parsed = JSON.parse(trimmed.slice(5).trim()); + + // Some upstream Responses-compatible providers leak an initial Chat Completions + // bootstrap chunk (assistant role + empty content) before emitting proper + // `response.*` events. That chunk is invalid on /v1/responses and breaks strict + // clients like OpenCode, so drop it only for Responses-native consumers. + const hasActiveDeltaValue = (value: unknown): boolean => { + if (typeof value === "string") return value.length > 0; + if (Array.isArray(value)) + return value.some((entry) => hasActiveDeltaValue(entry)); + if (value && typeof value === "object") { + return Object.values(value).some((entry) => hasActiveDeltaValue(entry)); + } + return value !== null && value !== undefined; + }; + + const isEmptyAssistantBootstrapChunkForResponsesClient = + ctx.clientExpectsResponsesStream && + parsed?.object === "chat.completion.chunk" && + Array.isArray(parsed?.choices) && + parsed.choices.length > 0 && + parsed.choices.every((choice) => { + const candidate = choice && typeof choice === "object" ? choice : {}; + const delta = + candidate.delta && typeof candidate.delta === "object" + ? candidate.delta + : null; + + if (!delta || delta.role !== "assistant") return false; + if (hasActiveDeltaValue(delta.content)) return false; + if (candidate.finish_reason !== null && candidate.finish_reason !== undefined) { + return false; + } + + const { role: _role, content: _content, ...restDelta } = delta; + return !hasActiveDeltaValue(restDelta); + }); + + if (isEmptyAssistantBootstrapChunkForResponsesClient) { + continue; + } + + // Detect Responses SSE payloads (have a `type` field like "response.created", + // "response.output_item.added", etc.) and skip Chat Completions-specific + // sanitization to avoid corrupting the stream for Responses-native clients. + const isResponsesSSE = + parsed.type && + typeof parsed.type === "string" && + parsed.type.startsWith("response."); + + // Detect Claude SSE payloads. Includes "ping" and "error" to ensure + // they bypass the Chat Completions sanitization path which would + // incorrectly process or drop them. + const isClaudeSSE = + parsed.type && + typeof parsed.type === "string" && + (parsed.type.startsWith("message") || + parsed.type.startsWith("content_block") || + parsed.type === "ping" || + parsed.type === "error"); + + if (isResponsesSSE) { + const responsesIdsNormalized = normalizeResponsesSseIds(parsed as JsonRecord); + const parsedResponse = + parsed.response && + typeof parsed.response === "object" && + !Array.isArray(parsed.response) + ? (parsed.response as JsonRecord) + : null; + const responseId = + (parsedResponse ? stringifyIdValue(parsedResponse.id) : null) || + stringifyIdValue(parsed.response_id); + if (responseId) { + ctx.passthroughResponsesId = responseId; + } + const extracted = extractUsage(parsed); + if (extracted) { + ctx.usage = extracted; + } + if (typeof parsed.delta === "string") { + ctx.totalContentLength += parsed.delta.length; + } + if ( + parsed.type === "response.output_text.delta" && + typeof parsed.delta === "string" + ) { + const incomingDelta = parsed.delta; + const bufferedCandidate = + ctx.passthroughBufferedTextualToolCallContent + incomingDelta; + if ( + ctx.passthroughBufferedTextualToolCallContent || + containsTextualToolCallCandidate(incomingDelta) + ) { + const parsedCandidate = parseTextualToolCallCandidate(bufferedCandidate); + if (parsedCandidate?.kind === "complete") { + const collectedToolCall = collectPassthroughTextualToolCall( + bufferedCandidate, + ctx.passthroughToolCalls, + ctx.allowedToolNames + ); + if (collectedToolCall) { + ctx.passthroughHasToolCalls = true; + const responseToolCallEvents = + buildResponsesFunctionCallEvents(collectedToolCall); + output = formatSSEDataEvents(responseToolCallEvents); + ctx.clientPayloadCollector.push(...responseToolCallEvents); + ctx.reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(ctx.encoder.encode(output)); + injectedUsage = true; + } else { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + ctx.passthroughBufferedTextualToolCallContent = ""; + parsed.delta = ""; + } else if (parsedCandidate?.kind === "partial") { + ctx.passthroughBufferedTextualToolCallContent = appendBoundedText( + ctx.passthroughBufferedTextualToolCallContent, + incomingDelta + ); + parsed.delta = ""; + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } else { + if (ctx.passthroughBufferedTextualToolCallContent) { + parsed.delta = + ctx.passthroughBufferedTextualToolCallContent + incomingDelta; + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + ctx.passthroughAccumulatedContent = appendBoundedText( + ctx.passthroughAccumulatedContent, + ctx.passthroughBufferedTextualToolCallContent + incomingDelta + ); + ctx.passthroughBufferedTextualToolCallContent = ""; + } + } else { + ctx.passthroughAccumulatedContent = appendBoundedText( + ctx.passthroughAccumulatedContent, + incomingDelta + ); + } + } + if (parsed.type === "response.failed") { + failurePayload = normalizeStreamFailurePayload(parsed); + } + if ( + parsed.type === "response.reasoning_summary_text.delta" || + parsed.type === "response.reasoning_summary_text.done" || + parsed.type === "response.reasoning_summary_part.done" + ) { + const reasoningKey = ctx.getResponsesReasoningKey(parsed); + if (reasoningKey) { + ctx.passthroughResponsesReasoningSummarySeen.add(reasoningKey); + } + } + if ( + parsed.type === "response.output_item.added" && + parsed.item?.type === "function_call" + ) { + const item = + parsed.item && typeof parsed.item === "object" && !Array.isArray(parsed.item) + ? { ...(parsed.item as JsonRecord) } + : null; + const pendingKey = + item && typeof item.id === "string" + ? item.id + : item && typeof item.call_id === "string" + ? item.call_id + : null; + if (item && pendingKey) { + if (typeof item.arguments !== "string") { + item.arguments = ""; + } + ctx.passthroughResponsesPendingFunctionCalls.set(pendingKey, item); + ctx.passthroughResponsesCurrentFunctionCallKey = pendingKey; + } + } + if (parsed.type === "response.function_call_arguments.delta") { + const pendingKey = + typeof parsed.item_id === "string" + ? parsed.item_id + : ctx.passthroughResponsesCurrentFunctionCallKey; + const pending = pendingKey + ? ctx.passthroughResponsesPendingFunctionCalls.get(pendingKey) + : undefined; + if (pending && typeof parsed.delta === "string") { + const previousArgs = + typeof pending.arguments === "string" ? pending.arguments : ""; + pending.arguments = previousArgs + parsed.delta; + } + } + if (parsed.type === "response.function_call_arguments.done") { + const pendingKey = + typeof parsed.item_id === "string" + ? parsed.item_id + : ctx.passthroughResponsesCurrentFunctionCallKey; + const pending = pendingKey + ? ctx.passthroughResponsesPendingFunctionCalls.get(pendingKey) + : undefined; + if (pending) { + if (typeof parsed.arguments === "string") { + pending.arguments = parsed.arguments; + } + pushUniqueResponsesOutputItems(ctx.passthroughResponsesOutputItems, [ + pending, + ]); + } + } + // Capture each completed output item so the final + // response.completed snapshot can be backfilled when upstream + // returns an empty `output` (happens with store: false). + if (parsed.type === "response.output_item.done" && parsed.item) { + const reasoningSummaryInjected = + ctx.ensureVisibleResponsesReasoningSummary(parsed); + ctx.emitSyntheticResponsesReasoningSummary(controller, parsed); + pushUniqueResponsesOutputItems(ctx.passthroughResponsesOutputItems, [ + parsed.item, + ]); + if (reasoningSummaryInjected) { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + if (parsed.item?.type === "function_call") { + const pendingKey = + typeof parsed.item.id === "string" + ? parsed.item.id + : typeof parsed.item.call_id === "string" + ? parsed.item.call_id + : null; + if (pendingKey) { + ctx.passthroughResponsesPendingFunctionCalls.delete(pendingKey); + if (ctx.passthroughResponsesCurrentFunctionCallKey === pendingKey) { + ctx.passthroughResponsesCurrentFunctionCallKey = null; + } + } + } + } + if ( + parsed.type === "response.completed" && + Array.isArray(parsed.response?.output) && + parsed.response.output.length > 0 + ) { + pushUniqueResponsesOutputItems( + ctx.passthroughResponsesOutputItems, + parsed.response.output + ); + } + if ( + parsed.type === "response.completed" && + ctx.passthroughResponsesPendingFunctionCalls.size > 0 + ) { + pushUniqueResponsesOutputItems(ctx.passthroughResponsesOutputItems, [ + ...ctx.passthroughResponsesPendingFunctionCalls.values(), + ]); + ctx.passthroughResponsesPendingFunctionCalls.clear(); + ctx.passthroughResponsesCurrentFunctionCallKey = null; + } + // Two transport-level fixes for Responses passthrough: + // 1) Strip echoed `instructions` + `tools` from lifecycle + // events — they can balloon a single SSE event past + // 100 KB and break parsers (e.g. GitHub Copilot CLI). + // 2) Backfill `response.completed.response.output` when + // upstream sent it empty (store: false) — some clients + // build their tool-call list from that snapshot rather + // than from per-item events. + const textualToolCallBackfilled = + parsed.type === "response.completed" && ctx.passthroughToolCalls.size > 0; + if (textualToolCallBackfilled) { + parsed = toResponsesCompletedWithToolCalls(parsed as JsonRecord, [ + ...ctx.passthroughToolCalls.values(), + ]) as typeof parsed; + } + const stripped = stripResponsesLifecycleEcho(parsed); + const backfilled = backfillResponsesCompletedOutput( + parsed, + ctx.passthroughResponsesOutputItems + ); + if ( + stripped || + backfilled || + textualToolCallBackfilled || + responsesIdsNormalized + ) { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + } else if (isClaudeSSE) { + // Claude SSE: extract ctx.usage, track content, forward as-is + const extracted = extractUsage(parsed); + if (extracted) { + // Non-destructive merge: never overwrite a positive value with 0 + // message_start carries input_tokens, message_delta carries output_tokens; + if (!ctx.usage) ctx.usage = {}; + const u = ctx.usage; + const eu = extracted as UsageTokenRecord; + if (eu.prompt_tokens > 0) u.prompt_tokens = eu.prompt_tokens; + if (eu.completion_tokens > 0) u.completion_tokens = eu.completion_tokens; + if (eu.total_tokens > 0) u.total_tokens = eu.total_tokens; + if (eu.cache_read_input_tokens) + u.cache_read_input_tokens = eu.cache_read_input_tokens; + if (eu.cache_creation_input_tokens) + u.cache_creation_input_tokens = eu.cache_creation_input_tokens; + } + if ( + shouldInjectClaudeEmptyResponseBeforeCurrentEvent( + ctx.claudeEmptyResponseLifecycle, + parsed + ) + ) { + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: true, + includeMessageDelta: + parsed.type === "message_stop" && + !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: false, + }); + } + updateClaudeEmptyResponseLifecycle(ctx.claudeEmptyResponseLifecycle, parsed); + const restoredToolName = restoreClaudePassthroughToolUseName( + parsed, + ctx.toolNameMap + ); + // Track content length and accumulate from Claude format + if (parsed.delta?.text) { + ctx.totalContentLength += parsed.delta.text.length; + ctx.passthroughAccumulatedContent = appendBoundedText( + ctx.passthroughAccumulatedContent, + parsed.delta.text + ); + } + if (parsed.delta?.thinking) { + ctx.totalContentLength += parsed.delta.thinking.length; + ctx.passthroughAccumulatedContent = appendBoundedText( + ctx.passthroughAccumulatedContent, + parsed.delta.thinking + ); + } + if (restoredToolName) { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + } else { + // ---------------------------------------------------- + // READ THIS !!! + // ---------------------------------------------------- + // IF YOU ARE AI OR AI AGENT DONT TOUCH THIS CODE !!! + // YOU WILL BREAK THIS CODE !!! + // GOOD LUCK + // ---------------------------------------------------- + // Chat Completions: full sanitization pipeline + + // OpenAI-compatible streaming with `stream_options.include_usage=true` + // ends with a ctx.usage-only chunk where `choices` is deliberately `[]`. + // Forward that standards-compliant chunk instead of turning it into an + // empty-response error. + // + // For a malformed empty `choices: []` chunk WITHOUT valid ctx.usage we DROP + // it (log server-side only). We must NOT inject an assistant-content + // chunk like "[OmniRoute] Upstream returned an empty response. Please + // retry." with finish_reason: "stop" — clients (Goose/opencode) feed that + // text back as a turn and spin in a retry loop. This restores the #3400 + // behavior that #3422 inadvertently reverted (regression #3388/#3502). + if (Array.isArray(parsed.choices) && parsed.choices.length === 0) { + const emptyChoicesUsage = extractUsage(parsed) ?? parsed.usage; + if (hasValidUsage(emptyChoicesUsage)) { + ctx.usage = emptyChoicesUsage; + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + clientPayload = parsed; + ctx.clientPayloadCollector.push(clientPayload); + ctx.reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(ctx.encoder.encode(output)); + continue; + } + + console.warn( + `[STREAM] Upstream returned empty choices array (${ctx.provider || "provider"}:${ctx.model || "unknown"}) — dropping chunk` + ); + continue; + } + + // Detect reasoning alias before sanitization strips it + const hadReasoningAlias = !!( + parsed.choices?.[0]?.delta?.reasoning && + typeof parsed.choices[0].delta.reasoning === "string" && + !parsed.choices[0].delta.reasoning_content + ); + const hadNonStringToolCallId = Array.isArray(parsed.choices) + ? parsed.choices.some( + (choice) => + Array.isArray(choice?.delta?.tool_calls) && + choice.delta.tool_calls.some( + (tc) => tc?.id != null && typeof tc.id !== "string" + ) + ) + : false; + const hadNonStringTopLevelId = + parsed?.id != null && typeof parsed.id !== "string"; + + parsed = sanitizeStreamingChunk(parsed); + if ( + parsed && + typeof parsed === "object" && + !Array.isArray(parsed) && + (parsed as Record)[OMIT_STREAMING_CHUNK_MARKER] === true + ) { + continue; + } + + const idFixed = hadNonStringTopLevelId ? false : fixInvalidId(parsed); + + if (!hasValuableContent(parsed, FORMATS.OPENAI)) { + continue; + } + + const delta = parsed.choices?.[0]?.delta; + let textualToolCallConverted = false; + let toolCallIdCoerced = false; + + // Extract tags from streaming content + if (delta?.content && typeof delta.content === "string") { + const { content, thinking } = extractThinkingFromContent(delta.content); + delta.content = content; + if (thinking && !delta.reasoning_content) { + delta.reasoning_content = thinking; + } + } + + // Split combined reasoning+content deltas into separate SSE events. + // Standard OpenAI streaming never mixes both fields in one delta; + // clients (e.g. LobeChat) may skip content when reasoning_content + // is present, causing the first content token to be lost. + if (delta?.reasoning_content && delta?.content) { + const reasoningChunk = JSON.parse(JSON.stringify(parsed)); + const rDelta = reasoningChunk.choices[0].delta; + delete rDelta.content; + reasoningChunk.choices[0].finish_reason = null; + delete reasoningChunk.usage; + const rOutput = `data: ${JSON.stringify(reasoningChunk)}\n`; + ctx.passthroughAccumulatedReasoning = appendBoundedText( + ctx.passthroughAccumulatedReasoning, + delta.reasoning_content + ); + ctx.totalContentLength += delta.reasoning_content.length; + ctx.clientPayloadCollector.push(reasoningChunk); + ctx.reqLogger?.appendConvertedChunk?.(rOutput); + controller.enqueue(ctx.encoder.encode(rOutput)); + controller.enqueue(ctx.encoder.encode("\n")); + delete delta.reasoning_content; + } + + // Track whether we need to re-serialize (separate from injectedUsage + // to avoid blocking subsequent finish_reason / ctx.usage mutations) + const needsReserialization = + hadReasoningAlias || (delta?.content === "" && delta?.reasoning_content); + + // T18: Track if we saw tool calls & accumulate for call log + if (delta?.tool_calls && delta.tool_calls.length > 0) { + ctx.passthroughHasToolCalls = true; + ctx.lastToolCallChunkTime = Date.now(); + for (const tc of delta.tool_calls) { + // Note: sanitizeStreamingChunk above already coerces non-string + // tool_call IDs, but this defensive check catches edge cases + // where sanitize didn't run (e.g. flush path shortcuts). + if (tc?.id != null && typeof tc.id !== "string") { + tc.id = String(tc.id); + toolCallIdCoerced = true; + } + // Key by index first — id only appears on the first delta in OpenAI streaming + let key: string; + if (Number.isInteger(tc?.index)) { + key = `idx:${tc.index}`; + } else if (tc?.id != null) { + key = `id:${tc.id}`; + } else { + key = `seq:${++ctx.passthroughToolCallSeq}`; + } + const existing = ctx.passthroughToolCalls.get(key); + const deltaArgs = + typeof tc?.function?.arguments === "string" ? tc.function.arguments : ""; + if (!existing) { + ctx.passthroughToolCalls.set(key, { + id: tc?.id != null ? String(tc.id) : null, + index: Number.isInteger(tc?.index) + ? tc.index + : ctx.passthroughToolCalls.size, + type: tc?.type || "function", + function: { + name: tc?.function?.name || "", + arguments: deltaArgs, + }, + }); + } else { + if (tc?.id) existing.id = existing.id || String(tc.id); + if (tc?.function?.name && !existing.function.name) + existing.function.name = tc.function.name; + existing.function.arguments += deltaArgs; + } + } + } + + const content = delta?.content || delta?.reasoning_content; + if (typeof content === "string") { + ctx.totalContentLength += content.length; + + if (!ctx.contentAfterToolSeen) { + const toolTs = ctx.toolFinishTime || ctx.pendingToolFinishTime; + const lastChunkTs = ctx.lastToolCallChunkTime; + if (toolTs || lastChunkTs) { + ctx.contentAfterToolSeen = true; + const now = Date.now(); + try { + recordToolLatency( + ctx.provider || "unknown", + toolTs ? now - toolTs : null, + lastChunkTs ? now - lastChunkTs : null + ); + } catch {} + ctx.pendingToolFinishTime = null; + } + } + } + { + const guarded = ctx.applyTextualToolCallStreamingGuard( + parsed as Record + ); + parsed = guarded.parsed as typeof parsed; + textualToolCallConverted = guarded.textualToolCallConverted; + } + if (typeof delta?.reasoning_content === "string") + ctx.passthroughAccumulatedReasoning = appendBoundedText( + ctx.passthroughAccumulatedReasoning, + delta.reasoning_content + ); + + const extracted = extractUsage(parsed); + if (extracted) { + ctx.usage = extracted; + } + + const isFinishChunk = parsed.choices?.[0]?.finish_reason; + + if (isFinishChunk && ctx.passthroughHasToolCalls) { + ctx.toolFinishTime = Date.now(); + try { + markToolFinish(ctx.sessionId); + } catch {} + } + + // T18: Normalize finish_reason to 'tool_calls' if tool calls were used + if ( + isFinishChunk && + ctx.passthroughHasToolCalls && + parsed.choices[0].finish_reason !== "tool_calls" + ) { + parsed.choices[0].finish_reason = "tool_calls"; + // If we modify it, we must output the modified object + if (!injectedUsage && hasValidUsage(parsed.usage)) { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + } + if ( + isFinishChunk && + !hasValidUsage(parsed.usage) && + !ctx.expectsOpenAIUsageOnlyChunk + ) { + const estimated = estimateUsage( + ctx.body, + ctx.totalContentLength, + FORMATS.OPENAI + ); + parsed.usage = filterUsageForFormat(estimated, FORMATS.OPENAI); + output = `data: ${JSON.stringify(parsed)}\n`; + ctx.usage = estimated; + injectedUsage = true; + } else if (isFinishChunk && ctx.usage) { + const buffered = addBufferToUsage(ctx.usage); + parsed.usage = filterUsageForFormat(buffered, FORMATS.OPENAI); + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } else if (textualToolCallConverted) { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } else if ( + idFixed || + needsReserialization || + toolCallIdCoerced || + hadNonStringToolCallId || + hadNonStringTopLevelId + ) { + output = `data: ${JSON.stringify(parsed)}\n`; + injectedUsage = true; + } + } + + clientPayload = parsed; + } catch {} + } + + if (!injectedUsage) { + if (line.startsWith("data:") && !line.startsWith("data: ")) { + output = "data: " + line.slice(5) + "\n"; + } else { + output = line + "\n"; + } + } + + if ( + !trimmed && + ctx.pendingPassthroughEventLine && + !ctx.pendingPassthroughEventEmitted + ) { + output = `${ctx.pendingPassthroughEventLine}\n${output}`; + ctx.pendingPassthroughEventEmitted = true; + } + + output = ctx.maybePrefixPendingPassthroughEvent(output, line); + + if (clientPayload) { + ctx.clientPayloadCollector.push(clientPayload); + } + + ctx.reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(ctx.encoder.encode(output)); + if (failurePayload) { + if (ctx.onFailure) { + try { + void ctx.onFailure(failurePayload); + } catch {} + } + ctx.clearIdleTimer(); + trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false); + controller.error( + markPendingRequestCleared(new Error(failurePayload.message || "Upstream failure")) + ); + return; + } + if (!trimmed) { + ctx.clearPendingPassthroughEvent(); + } + continue; + } + + // Translate ctx.mode + if (!trimmed) continue; + + if (ctx.state?.upstreamError) { + continue; + } + + const parsed = parseSSELine(trimmed); + if (!parsed) continue; + ctx.providerPayloadCollector.push(parsed); + + if (parsed && parsed.done) { + continue; + } + + if (parsed.choices?.[0]?.delta?.tool_calls) { + ctx.lastToolCallChunkTime = Date.now(); + } + if (parsed.choices?.[0]?.finish_reason === "tool_calls") { + ctx.toolFinishTime = Date.now(); + try { + markToolFinish(ctx.sessionId); + } catch {} + } + + // Track content length and accumulate for call log (from raw ctx.provider chunk, so content is never missed) + // Do this before translation so we capture content regardless of translator output shape + + // Claude format + if (parsed.delta?.text) { + const t = parsed.delta.text; + ctx.totalContentLength += t.length; + if (ctx.state?.accumulatedContent !== undefined && typeof t === "string") + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, t); + } + if (parsed.delta?.thinking) { + const t = parsed.delta.thinking; + ctx.totalContentLength += t.length; + if (ctx.state?.accumulatedContent !== undefined && typeof t === "string") + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, t); + } + + // OpenAI format + if (parsed.choices?.[0]?.delta?.content) { + const c = parsed.choices[0].delta.content; + if (typeof c === "string") { + ctx.totalContentLength += c.length; + if (ctx.state?.accumulatedContent !== undefined) + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, c); + } else if (Array.isArray(c)) { + for (const part of c) { + if (part?.text && typeof part.text === "string") { + ctx.totalContentLength += part.text.length; + if (ctx.state?.accumulatedContent !== undefined) + ctx.state.accumulatedContent = appendBoundedText( + ctx.state.accumulatedContent, + part.text + ); + } + } + } + } + if (parsed.choices?.[0]?.delta?.reasoning_content) { + const r = parsed.choices[0].delta.reasoning_content; + if (typeof r === "string") { + ctx.totalContentLength += r.length; + if (ctx.state?.accumulatedContent !== undefined) + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, r); + } + } + // Normalize `reasoning` alias → `reasoning_content` (NVIDIA kimi-k2.5 etc.) + if ( + parsed.choices?.[0]?.delta?.reasoning && + !parsed.choices?.[0]?.delta?.reasoning_content + ) { + const r = parsed.choices[0].delta.reasoning; + if (typeof r === "string") { + parsed.choices[0].delta.reasoning_content = r; + delete parsed.choices[0].delta.reasoning; + ctx.totalContentLength += r.length; + if (ctx.state?.accumulatedContent !== undefined) + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, r); + } + } + + // Gemini / Cloud Code format - may have multiple parts + // Cloud Code API wraps in { response: { candidates: [...] } }, so unwrap. + // Only applies to Gemini-family formats — skip for OpenAI, Claude, etc. + const isGeminiFormat = + ctx.targetFormat === FORMATS.GEMINI || + ctx.targetFormat === FORMATS.GEMINI_CLI || + ctx.targetFormat === FORMATS.ANTIGRAVITY; + const geminiChunk = isGeminiFormat ? unwrapGeminiChunk(parsed) : parsed; + if (geminiChunk.candidates?.[0]?.content?.parts) { + for (const part of geminiChunk.candidates[0].content.parts) { + if (part.text && typeof part.text === "string") { + ctx.totalContentLength += part.text.length; + if (ctx.state?.accumulatedContent !== undefined) + ctx.state.accumulatedContent = appendBoundedText( + ctx.state.accumulatedContent, + part.text + ); + } + } + } + + // Generic fallback: delta string, top-level content/text (e.g. some SSE payloads) + if (ctx.state?.accumulatedContent !== undefined) { + if (typeof (parsed as JsonRecord).delta === "string") { + const d = (parsed as JsonRecord).delta as string; + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, d); + ctx.totalContentLength += d.length; + } + if (typeof (parsed as JsonRecord).content === "string") { + const c = (parsed as JsonRecord).content as string; + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, c); + ctx.totalContentLength += c.length; + } + if (typeof (parsed as JsonRecord).text === "string") { + const t = (parsed as JsonRecord).text as string; + ctx.state.accumulatedContent = appendBoundedText(ctx.state.accumulatedContent, t); + ctx.totalContentLength += t.length; + } + } + + const translateHasContent = + typeof parsed.delta?.text === "string" || + typeof parsed.choices?.[0]?.delta?.content === "string" || + typeof parsed.choices?.[0]?.delta?.reasoning_content === "string"; + if (translateHasContent && !ctx.contentAfterToolSeen) { + const toolTs = ctx.toolFinishTime || ctx.pendingToolFinishTime; + const lastChunkTs = ctx.lastToolCallChunkTime; + if (toolTs || lastChunkTs) { + ctx.contentAfterToolSeen = true; + const now = Date.now(); + try { + recordToolLatency( + ctx.provider || "unknown", + toolTs ? now - toolTs : null, + lastChunkTs ? now - lastChunkTs : null + ); + } catch {} + ctx.pendingToolFinishTime = null; + } + } + + // Extract ctx.usage + const extracted = extractUsage(parsed); + if (extracted) ctx.state.usage = extracted; // Keep original ctx.usage for logging + + // Translate: ctx.targetFormat -> openai -> ctx.sourceFormat + const translated = translateResponse( + ctx.targetFormat, + ctx.sourceFormat, + parsed, + ctx.state + ); + + // Log OpenAI intermediate chunks (if available) + for (const item of getOpenAIIntermediateChunks(translated)) { + const openaiOutput = formatSSE(item, FORMATS.OPENAI); + ctx.reqLogger?.appendOpenAIChunk?.(openaiOutput); + } + + if (translated?.length > 0) { + for (const item of translated) { + ctx.emitTranslatedClientItem(controller, item); + } + } + } + }, + + async flush(controller) { + // Clean up idle watchdog timer + if (ctx.idleTimer) { + ctx.clearIdleTimer(); + } + if (ctx.streamTimedOut) { + return; + } + trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false); + try { + const remaining = ctx.decoder.decode(); + if (remaining) ctx.buffer += remaining; + + if (ctx.mode === STREAM_MODE.PASSTHROUGH) { + const bufferedLine = ctx.buffer.trim(); + if (ctx.skipPassthroughEvent || /^event:\s*keepalive\b/i.test(bufferedLine)) { + ctx.skipPassthroughEvent = false; + ctx.clearPendingPassthroughEvent(); + } else if (ctx.buffer) { + let output = ctx.buffer; + if (ctx.buffer.startsWith("data:") && !ctx.buffer.startsWith("data: ")) { + output = "data: " + ctx.buffer.slice(5); + } + const bufferedPayload = parseSSELine(bufferedLine); + if (bufferedPayload) { + ctx.providerPayloadCollector.push(bufferedPayload); + if ( + shouldInjectClaudeEmptyResponseBeforeCurrentEvent( + ctx.claudeEmptyResponseLifecycle, + bufferedPayload + ) + ) { + const eventType = getClaudeEventType(bufferedPayload); + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: true, + includeMessageDelta: + eventType === "message_stop" && + !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: false, + }); + } + if (isClaudeEventPayload(bufferedPayload)) { + updateClaudeEmptyResponseLifecycle( + ctx.claudeEmptyResponseLifecycle, + bufferedPayload + ); + } + ctx.clientPayloadCollector.push(bufferedPayload); + + // Normalize numeric IDs for final buffered data: chunk (same as transform path) + if (typeof bufferedPayload === "object" && !Array.isArray(bufferedPayload)) { + const flushedParsed = bufferedPayload as JsonRecord; + const flushedType = + typeof flushedParsed.type === "string" ? flushedParsed.type : ""; + const isResponses = flushedType.startsWith("response."); + const isClaude = isClaudeEventPayload(flushedParsed); + if (isResponses) { + if (normalizeResponsesSseIds(flushedParsed)) { + output = `data: ${JSON.stringify(flushedParsed)}\n`; + } + } else if (!isClaude) { + let flushChanged = false; + const flushedHadNonStringTopLevelId = + flushedParsed?.id != null && typeof flushedParsed.id !== "string"; + if (flushedHadNonStringTopLevelId) { + flushedParsed.id = String(flushedParsed.id); + flushChanged = true; + } + if (Array.isArray(flushedParsed.choices)) { + for (const choice of flushedParsed.choices as JsonRecord[]) { + const tcs = (choice as JsonRecord | undefined)?.delta as + | JsonRecord + | undefined; + if (Array.isArray(tcs?.tool_calls)) { + for (const tc of tcs.tool_calls as JsonRecord[]) { + if (tc?.id != null && typeof tc.id !== "string") { + tc.id = String(tc.id); + flushChanged = true; + } + } + } + } + } + if (flushChanged) { + output = `data: ${JSON.stringify(flushedParsed)}\n`; + } + } + } + } + if ( + !bufferedLine && + ctx.pendingPassthroughEventLine && + !ctx.pendingPassthroughEventEmitted + ) { + output = `${ctx.pendingPassthroughEventLine}\n${output}`; + ctx.pendingPassthroughEventEmitted = true; + } + output = ctx.maybePrefixPendingPassthroughEvent(output, ctx.buffer); + ctx.reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(ctx.encoder.encode(output)); + } + + if (shouldInjectClaudeEmptyResponseOnFlush(ctx.claudeEmptyResponseLifecycle)) { + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: true, + includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop, + }); + } else if ( + shouldInjectClaudeMissingFinalizersOnFlush(ctx.claudeEmptyResponseLifecycle) + ) { + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: false, + includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop, + }); + } + ctx.clearPendingPassthroughEvent(); + + if (ctx.passthroughBufferedTextualToolCallContent) { + // Flush any remaining buffered content as plain text. + // Previously gated on !includes("Arguments:"), which silently dropped + // incomplete tool-call headers (ctx.buffer held "Arguments:" but JSON was + // never finished before stream ended) — fix #3355 bug 2. + let flushOutput = ""; + if (ctx.clientExpectsResponsesStream) { + const syntheticChunk = { + type: "response.output_text.delta", + delta: ctx.passthroughBufferedTextualToolCallContent, + }; + flushOutput = `data: ${JSON.stringify(syntheticChunk)}\n\n`; + } else if (ctx.clientExpectsClaudeStream) { + const syntheticChunk = { + type: "content_block_delta", + index: 0, + delta: { + type: "text_delta", + text: ctx.passthroughBufferedTextualToolCallContent, + }, + }; + flushOutput = `data: ${JSON.stringify(syntheticChunk)}\n\n`; + } else { + const syntheticChunk = { + id: ctx.passthroughResponsesId || `chatcmpl-${Date.now()}`, + object: "chat.completion.chunk", + created: Math.floor(Date.now() / 1000), + model: ctx.model || "unknown", + choices: [ + { + index: 0, + delta: { + content: ctx.passthroughBufferedTextualToolCallContent, + }, + finish_reason: null, + }, + ], + }; + flushOutput = `data: ${JSON.stringify(syntheticChunk)}\n\n`; + } + ctx.reqLogger?.appendConvertedChunk?.(flushOutput); + controller.enqueue(ctx.encoder.encode(flushOutput)); + ctx.passthroughAccumulatedContent = appendBoundedText( + ctx.passthroughAccumulatedContent, + ctx.passthroughBufferedTextualToolCallContent + ); + ctx.passthroughBufferedTextualToolCallContent = ""; + } + + // Estimate ctx.usage if ctx.provider didn't return valid ctx.usage + if (!hasValidUsage(ctx.usage) && ctx.totalContentLength > 0) { + ctx.usage = estimateUsage( + ctx.body, + ctx.totalContentLength, + ctx.sourceFormat || FORMATS.OPENAI + ); + } + + if (hasValidUsage(ctx.usage)) { + logUsage(ctx.provider, ctx.usage, ctx.model, ctx.connectionId, ctx.apiKeyInfo); + } else { + appendRequestLog({ + model: ctx.model, + provider: ctx.provider, + connectionId: ctx.connectionId, + tokens: null, + status: "200 OK", + }).catch(() => {}); + } + if (!ctx.doneSent) { + await ctx.emitFinalSseMetadata(controller, ctx.usage); + ctx.doneSent = true; + if (ctx.shouldEmitDoneTerminator) { + ctx.clientPayloadCollector.push({ done: true }); + const doneOutput = "data: [DONE]\n\n"; + ctx.reqLogger?.appendConvertedChunk?.(doneOutput); + controller.enqueue(ctx.encoder.encode(doneOutput)); + } + } + // Notify caller for call log persistence (include full response ctx.body with accumulated content) + if (ctx.onComplete) { + try { + const u = ctx.usage as Record | null; + const prompt = Number(u?.prompt_tokens ?? u?.input_tokens ?? 0); + const completion = Number(u?.completion_tokens ?? u?.output_tokens ?? 0); + let content = ctx.passthroughAccumulatedContent.trim() || ""; + const finalBufferedTextualToolCall = + ctx.passthroughBufferedTextualToolCallContent.trim(); + if (finalBufferedTextualToolCall) { + if ( + collectPassthroughTextualToolCall( + finalBufferedTextualToolCall, + ctx.passthroughToolCalls, + ctx.allowedToolNames + ) + ) { + ctx.passthroughHasToolCalls = true; + } + ctx.passthroughBufferedTextualToolCallContent = ""; + } + if ( + content && + collectPassthroughTextualToolCall( + content, + ctx.passthroughToolCalls, + ctx.allowedToolNames + ) + ) { + ctx.passthroughHasToolCalls = true; + content = ""; + } else if (containsMalformedTextualToolCall(content, ctx.allowedToolNames)) { + content = ""; + } + const message: Record = { + role: "assistant", + content: content || null, + }; + const reasoning = ctx.passthroughAccumulatedReasoning.trim(); + if (reasoning) { + message.reasoning_content = reasoning; + } + if (ctx.passthroughToolCalls.size > 0) { + message.tool_calls = [...ctx.passthroughToolCalls.values()].sort( + (a, b) => a.index - b.index + ); + } + // Hardening: log empty assistant response after tool completion + // for observability — helps diagnose Copilot "Sorry, no response was returned" + if (ctx.passthroughHasToolCalls && !content.trim() && !reasoning.trim()) { + console.warn( + `[STREAM] Empty assistant response after tool_calls completion (${ctx.provider || "provider"}:${ctx.model || "unknown"}) — ctx.sessionId=${ctx.sessionId}` + ); + } + + const responseBody = { + choices: [ + { + message, + finish_reason: ctx.passthroughHasToolCalls ? "tool_calls" : "stop", + }, + ], + usage: { + prompt_tokens: prompt, + completion_tokens: completion, + total_tokens: prompt + completion, + }, + _streamed: true, + }; + ctx.onComplete({ + status: 200, + usage: ctx.usage, + responseBody, + providerPayload: ctx.providerPayloadCollector.build( + buildStreamSummaryFromEvents( + ctx.providerPayloadCollector.getEvents(), + ctx.sourceFormat, + ctx.model + ), + { includeEvents: false } + ), + clientPayload: ctx.clientPayloadCollector.build(responseBody, { + includeEvents: false, + }), + }); + } catch {} + } + return; + } + + // Translate ctx.mode: process remaining ctx.buffer + if (ctx.buffer.trim()) { + const parsed = parseSSELine(ctx.buffer.trim()); + if (parsed && !parsed.done) { + ctx.providerPayloadCollector.push(parsed); + // Extract ctx.usage from remaining ctx.buffer — if the ctx.usage-bearing event + // (e.g. response.completed) is the last SSE line, it ends up here + // in the flush handler where extractUsage was not called. + // Non-destructive merge: some providers send ctx.usage across multiple + // events (e.g. prompt_tokens in message_start, completion_tokens + // in message_delta). Direct assignment would lose earlier data. + const extracted = extractUsage(parsed); + if (extracted) { + if (!ctx.state.usage) { + ctx.state.usage = extracted; + } else { + const su = ctx.state.usage as Record; + const eu = extracted as Record; + if (eu.prompt_tokens > 0) su.prompt_tokens = eu.prompt_tokens; + if (eu.completion_tokens > 0) su.completion_tokens = eu.completion_tokens; + if (eu.total_tokens > 0) su.total_tokens = eu.total_tokens; + if (eu.cache_read_input_tokens > 0) + su.cache_read_input_tokens = eu.cache_read_input_tokens; + if (eu.cache_creation_input_tokens > 0) + su.cache_creation_input_tokens = eu.cache_creation_input_tokens; + if (eu.cached_tokens > 0) su.cached_tokens = eu.cached_tokens; + if (eu.reasoning_tokens > 0) su.reasoning_tokens = eu.reasoning_tokens; + } + } + + const translated = translateResponse( + ctx.targetFormat, + ctx.sourceFormat, + parsed, + ctx.state + ); + + // Log OpenAI intermediate chunks + for (const item of getOpenAIIntermediateChunks(translated)) { + const openaiOutput = formatSSE(item, FORMATS.OPENAI); + ctx.reqLogger?.appendOpenAIChunk?.(openaiOutput); + } + + if (translated?.length > 0) { + for (const item of translated) { + ctx.emitTranslatedClientItem(controller, item); + } + } + } + } + + if (ctx.state?.upstreamError) { + const err = ctx.state.upstreamError; + trackPendingRequest(ctx.model, ctx.provider, ctx.connectionId, false); + if (ctx.onFailure) { + try { + void ctx.onFailure({ + status: err.status, + message: err.message, + code: err.code, + type: err.type, + }); + } catch {} + } + + const errorBody = buildErrorBody(err.status, err.message); + if (ctx.onComplete) { + try { + ctx.onComplete({ + status: err.status, + usage: ctx.state?.usage, + responseBody: errorBody, + providerPayload: ctx.providerPayloadCollector.build( + buildStreamSummaryFromEvents( + ctx.providerPayloadCollector.getEvents(), + ctx.targetFormat, + ctx.model + ), + { includeEvents: false } + ), + clientPayload: ctx.clientPayloadCollector.build(errorBody, { + includeEvents: false, + }), + }); + } catch {} + } + + ctx.clearIdleTimer(); + controller.error( + markPendingRequestCleared(new Error(err.message || "Upstream failure")) + ); + return; + } + + // Flush remaining events (only once at stream end) + const flushed = translateResponse(ctx.targetFormat, ctx.sourceFormat, null, ctx.state); + + // Log OpenAI intermediate chunks for flushed events + for (const item of getOpenAIIntermediateChunks(flushed)) { + const openaiOutput = formatSSE(item, FORMATS.OPENAI); + ctx.reqLogger?.appendOpenAIChunk?.(openaiOutput); + } + + if (flushed?.length > 0) { + for (const item of flushed) { + ctx.emitTranslatedClientItem(controller, item); + } + } + + if (ctx.sourceFormat === FORMATS.CLAUDE) { + if (shouldInjectClaudeEmptyResponseOnFlush(ctx.claudeEmptyResponseLifecycle)) { + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: true, + includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop, + }); + } else if ( + shouldInjectClaudeMissingFinalizersOnFlush(ctx.claudeEmptyResponseLifecycle) + ) { + ctx.emitSyntheticClaudeEmptyResponse(controller, { + includeContentBlock: false, + includeMessageDelta: !ctx.claudeEmptyResponseLifecycle.hasMessageDelta, + includeMessageStop: !ctx.claudeEmptyResponseLifecycle.hasMessageStop, + }); + } + } + + /** + * Usage injection strategy: + * Usage data (input/output tokens) is injected into the last content chunk + * or the finish_reason chunk rather than sent as a separate SSE event. + * This ensures all major clients (Claude CLI, Continue, Cursor) receive + * ctx.usage data even if they stop reading after the finish signal. + * The ctx.usage ctx.buffer (state.usage) accumulates across chunks and is only + * emitted once at stream end when merged into the final translated chunk. + */ + + // Send [DONE] (only if not already sent during transform) + if (!ctx.doneSent) { + await ctx.emitFinalSseMetadata( + controller, + ctx.state?.usage as Record | null + ); + ctx.doneSent = true; + if (ctx.shouldEmitDoneTerminator) { + ctx.clientPayloadCollector.push({ done: true }); + const doneOutput = "data: [DONE]\n\n"; + ctx.reqLogger?.appendConvertedChunk?.(doneOutput); + controller.enqueue(ctx.encoder.encode(doneOutput)); + } + } + + // Estimate ctx.usage if ctx.provider didn't return valid ctx.usage (for translate ctx.mode) + if (!hasValidUsage(ctx.state?.usage) && ctx.totalContentLength > 0) { + ctx.state.usage = estimateUsage(ctx.body, ctx.totalContentLength, ctx.sourceFormat); + } + + if (hasValidUsage(ctx.state?.usage)) { + logUsage( + ctx.state?.provider || ctx.targetFormat, + ctx.state.usage, + ctx.model, + ctx.connectionId, + ctx.apiKeyInfo + ); + } else { + appendRequestLog({ + model: ctx.model, + provider: ctx.provider, + connectionId: ctx.connectionId, + tokens: null, + status: "200 OK", + }).catch(() => {}); + } + // Notify caller for call log persistence (include full response ctx.body with accumulated content) + if (ctx.onComplete) { + try { + const u = ctx.state?.usage as Record | null | undefined; + const prompt = Number(u?.prompt_tokens ?? u?.input_tokens ?? 0); + const completion = Number(u?.completion_tokens ?? u?.output_tokens ?? 0); + let content = (ctx.state?.accumulatedContent ?? "").trim() || ""; + const normalizedToolCalls: ToolCall[] = ctx.state?.toolCalls?.size + ? [...ctx.state.toolCalls.values()] + .map( + (tc: Record): ToolCall => ({ + id: tc.id != null ? String(tc.id) : null, + index: (tc.index as number) ?? (tc.blockIndex as number) ?? 0, + type: (tc.type as string) ?? "function", + function: (tc.function as ToolCall["function"]) ?? { + name: (tc.name as string) ?? "", + arguments: "", + }, + }) + ) + .sort((a, b) => a.index - b.index) + : []; + const textualToolCall = parseTextualToolCallFromContent(content); + if (textualToolCall) { + normalizedToolCalls.push({ + id: `call_${Date.now()}_${normalizedToolCalls.length}`, + index: normalizedToolCalls.length, + type: "function", + function: { + name: textualToolCall.name, + arguments: JSON.stringify(textualToolCall.args || {}), + }, + }); + content = ""; + } else if (containsMalformedTextualToolCall(content, ctx.allowedToolNames)) { + content = ""; + } + const message: Record = { + role: "assistant", + content: content || null, + }; + const hasToolCalls = normalizedToolCalls.length > 0; + if (hasToolCalls) { + message.tool_calls = normalizedToolCalls; + } + const responseBody = { + choices: [ + { + message, + finish_reason: hasToolCalls ? "tool_calls" : "stop", + }, + ], + usage: { + prompt_tokens: prompt, + completion_tokens: completion, + total_tokens: prompt + completion, + }, + _streamed: true, + }; + ctx.onComplete({ + status: 200, + usage: ctx.state?.usage, + responseBody, + providerPayload: ctx.providerPayloadCollector.build( + buildStreamSummaryFromEvents( + ctx.providerPayloadCollector.getEvents(), + ctx.targetFormat, + ctx.model + ), + { includeEvents: false } + ), + clientPayload: ctx.clientPayloadCollector.build(responseBody, { + includeEvents: false, + }), + }); + } catch {} + } + } catch (error) { + console.log( + `[STREAM] Error in flush (${ctx.model || "unknown"}):`, + error.message || error + ); + } + }, + cancel(reason) { + ctx.clearIdleTimer(); + }, + }, + { highWaterMark: 16384 }, + { highWaterMark: 16384 } + ); +} + +// Convenience functions for backward compatibility +export function createSSETransformStreamWithLogger( + targetFormat: string, + sourceFormat: string, + provider: string | null = null, + reqLogger: StreamLogger | null = null, + toolNameMap: unknown = null, + model: string | null = null, + connectionId: string | null = null, + body: unknown = null, + onComplete: ((payload: StreamCompletePayload) => void) | null = null, + apiKeyInfo: unknown = null, + onFailure: ((payload: StreamFailurePayload) => void | Promise) | null = null, + copilotCompatibleReasoning = false +) { + return createSSEStream({ + mode: STREAM_MODE.TRANSLATE, + targetFormat, + sourceFormat, + provider, + reqLogger, + toolNameMap, + model, + connectionId, + apiKeyInfo, + body, + onComplete, + onFailure, + copilotCompatibleReasoning, + }); +} + +export function createPassthroughStreamWithLogger( + provider: string | null = null, + reqLogger: StreamLogger | null = null, + toolNameMap: unknown = null, + model: string | null = null, + connectionId: string | null = null, + body: unknown = null, + onComplete: ((payload: StreamCompletePayload) => void) | null = null, + apiKeyInfo: unknown = null, + onFailure: ((payload: StreamFailurePayload) => void | Promise) | null = null, + clientResponseFormat: string | null = null +) { + return createSSEStream({ + mode: STREAM_MODE.PASSTHROUGH, + provider, + reqLogger, + toolNameMap, + model, + connectionId, + apiKeyInfo, + body, + onComplete, + onFailure, + clientResponseFormat, + }); +} diff --git a/open-sse/utils/stream/textualToolCalls.ts b/open-sse/utils/stream/textualToolCalls.ts new file mode 100644 index 00000000000..275cc7e8563 --- /dev/null +++ b/open-sse/utils/stream/textualToolCalls.ts @@ -0,0 +1,85 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { asRecord } from "./utils.ts"; +import { JsonRecord, ToolCall } from "./types.ts"; + +export function parseTextualToolCallFromContent(text: unknown): { name: string; args: unknown } | null { + const candidate = parseTextualToolCallCandidate(text); + return candidate?.kind === "complete" ? { name: candidate.name, args: candidate.args } : null; +} + +export function containsTextualToolCallCandidate(text: unknown): boolean { + return parseTextualToolCallCandidate(text) !== null; +} + +export function containsMalformedTextualToolCall( + text: unknown, + allowedToolNames?: Set | null +): boolean { + if (typeof text !== "string") return false; + const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, ""); + + let searchIdx = 0; + while (true) { + const idx = normalized.indexOf("[Tool call:", searchIdx); + if (idx === -1) break; + + const candidate = normalized.slice(idx); + if (isValidToolCallHeaderPrefix(candidate)) { + const parsed = parseTextualToolCallFromContent(candidate); + if (parsed) { + if (allowedToolNames?.size && !allowedToolNames.has(parsed.name)) { + return true; + } + } else { + return true; + } + } + + searchIdx = idx + 1; + } + return false; +} + +export function extractAllowedToolNames(body: unknown): Set | null { + const record = asRecord(body); + const tools = record.tools; + if (!Array.isArray(tools)) return null; + const names = new Set(); + for (const tool of tools) { + if (!tool || typeof tool !== "object" || Array.isArray(tool)) continue; + const item = tool as JsonRecord; + const directName = typeof item.name === "string" ? item.name.trim() : ""; + const fn = + item.function && typeof item.function === "object" && !Array.isArray(item.function) + ? (item.function as JsonRecord) + : null; + const functionName = typeof fn?.name === "string" ? fn.name.trim() : ""; + const name = functionName || directName; + if (name) names.add(name); + } + return names.size > 0 ? names : null; +} + +export function collectPassthroughTextualToolCall( + text: string, + toolCalls: Map, + allowedToolNames?: Set | null +): ToolCall | null { + const parsed = parseTextualToolCallFromContent(text); + if (!parsed) return null; + if (allowedToolNames?.size && !allowedToolNames.has(parsed.name)) return null; + const key = `textual:${toolCalls.size}`; + const toolCall: ToolCall = { + id: `call_${Date.now()}_${toolCalls.size}`, + index: toolCalls.size, + type: "function", + function: { + name: parsed.name, + arguments: JSON.stringify(parsed.args || {}), + }, + }; + toolCalls.set(key, toolCall); + return toolCall; +} \ No newline at end of file diff --git a/open-sse/utils/stream/types.ts b/open-sse/utils/stream/types.ts new file mode 100644 index 00000000000..728b303532e --- /dev/null +++ b/open-sse/utils/stream/types.ts @@ -0,0 +1,138 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +export type JsonRecord = Record; + +export type StreamLogger = { + appendProviderChunk?: (value: string) => void; + appendConvertedChunk?: (value: string) => void; + appendOpenAIChunk?: (value: string) => void; +}; + +export type StreamCompletePayload = { + status: number; + usage: unknown; + /** Minimal response body for call log (streaming: usage + note; non-streaming not used) */ + responseBody?: unknown; + providerPayload?: unknown; + clientPayload?: unknown; +}; + +export type StreamFailurePayload = { + status: number; + message: string; + code?: string; + type?: string; +}; + +export type StreamOptions = { + mode?: string; + targetFormat?: string; + sourceFormat?: string; + clientResponseFormat?: string | null; + copilotCompatibleReasoning?: boolean; + provider?: string | null; + reqLogger?: StreamLogger | null; + toolNameMap?: unknown; + model?: string | null; + connectionId?: string | null; + apiKeyInfo?: unknown; + body?: unknown; + onComplete?: ((payload: StreamCompletePayload) => void) | null; + onFailure?: ((payload: StreamFailurePayload) => void | Promise) | null; +}; + +export type TranslateState = ReturnType & { + provider?: string | null; + toolNameMap?: unknown; + signatureNamespace?: string | null; + usage?: unknown; + finishReason?: unknown; + copilotCompatibleReasoning?: boolean; + /** Accumulated message content for call log response body */ + accumulatedContent?: string; + upstreamError?: { + status: number; + type: string; + code: string; + message: string; + } | null; +}; + +export type ToolCall = { + id: string | null; + index: number; + type: string; + function: { name: string; arguments: string }; +}; + +export type UsageTokenRecord = Record; + +export type SSEStreamContext = { + mode: string; + targetFormat?: string; + sourceFormat?: string; + clientResponseFormat: string | null; + copilotCompatibleReasoning: boolean; + provider: string | null; + reqLogger: StreamLogger | null; + toolNameMap: unknown; + model: string | null; + connectionId: string | null; + apiKeyInfo: unknown; + body: unknown; + onComplete: ((payload: StreamCompletePayload) => void) | null; + onFailure: ((payload: StreamFailurePayload) => void | Promise) | null; + + clientExpectsResponsesStream: boolean; + clientExpectsClaudeStream: boolean; + shouldEmitDoneTerminator: boolean; + expectsOpenAIUsageOnlyChunk: boolean; + signatureNamespace: string | null; + + buffer: string; + usage: UsageTokenRecord | null; + passthroughHasToolCalls: boolean; + passthroughToolCalls: Map; + passthroughToolCallSeq: number; + allowedToolNames: string[]; + skipPassthroughEvent: boolean; + state: TranslateState | null; + totalContentLength: number; + passthroughAccumulatedContent: string; + passthroughAccumulatedReasoning: string; + passthroughBufferedTextualToolCallContent: string; + passthroughResponsesOutputItems: unknown[]; + passthroughResponsesPendingFunctionCalls: Map; + passthroughResponsesId: string | null; + passthroughResponsesCurrentFunctionCallKey: string | null; + passthroughResponsesReasoningSummarySeen: Set; + streamStartedAt: number; + lastToolCallChunkTime: number | null; + toolFinishTime: number | null; + contentAfterToolSeen: boolean; + sessionId: string; + pendingToolFinishTime: number | null; + doneSent: boolean; + pendingPassthroughEventLine: string | null; + pendingPassthroughEventEmitted: boolean; + lastChunkTime: number; + streamTimedOut: boolean; + + decoder: TextDecoder; + encoder: TextEncoder; + idleTimer: ReturnType | null; + claudeEmptyResponseLifecycle: Record; + providerPayloadCollector: { + push: (v: unknown) => void; + build: (...args: unknown[]) => unknown; + getEvents: () => unknown[]; + }; + clientPayloadCollector: { + push: (v: unknown) => void; + build: (...args: unknown[]) => unknown; + getEvents: () => unknown[]; + }; + requestRecord: JsonRecord; + requestStreamOptions: JsonRecord; +}; diff --git a/open-sse/utils/stream/utils.ts b/open-sse/utils/stream/utils.ts new file mode 100644 index 00000000000..ac58a57aea7 --- /dev/null +++ b/open-sse/utils/stream/utils.ts @@ -0,0 +1,54 @@ +import { convertOpenAIToResponsesToolCall } from "../handlers/responseTranslator.ts"; +import { v4 as uuidv4 } from "uuid"; + +import { createSSEStream } from "./streamCore.ts"; +import { JsonRecord } from "./types.ts"; + +/** + * Race a response body read against a timeout. + * Prevents indefinite hangs when the upstream sends headers but stalls on the body. + */ +export function withBodyTimeout( + promise: Promise, + timeoutMs: number = FETCH_BODY_TIMEOUT_MS +): Promise { + if (timeoutMs <= 0) return promise; + let timer: ReturnType; + const timeout = new Promise((_, reject) => { + timer = setTimeout(() => { + const err = new Error(`Response body read timeout after ${timeoutMs}ms`); + err.name = "BodyTimeoutError"; + reject(err); + }, timeoutMs); + }); + return Promise.race([promise, timeout]).finally(() => clearTimeout(timer)) as Promise; +} + +export function stringifyIdValue(value: unknown): string | null { + return value === null || value === undefined ? null : String(value); +} + +export function asRecord(value: unknown): JsonRecord { + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; +} + +export const STREAM_SUMMARY_TEXT_LIMIT = 64 * 1024; + +export function appendBoundedText(current: string, next: string): string { + if (!next) return current; + const combined = current + next; + if (combined.length <= STREAM_SUMMARY_TEXT_LIMIT) return combined; + return combined.slice(-STREAM_SUMMARY_TEXT_LIMIT); +} + +// Note: TextDecoder/TextEncoder are created per-stream inside createSSEStream() +// to avoid shared state issues with concurrent streams (TextDecoder with {stream:true} +// maintains internal buffering state between decode() calls). + +/** + * Stream modes + */ +export const STREAM_MODE = { + TRANSLATE: "translate", // Full translation between formats + PASSTHROUGH: "passthrough", // No translation, normalize output, extract usage +}; \ No newline at end of file diff --git a/package-lock.json b/package-lock.json index e053269520d..8176e182a1e 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "omniroute", - "version": "3.8.25", + "version": "3.8.26", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "omniroute", - "version": "3.8.25", + "version": "3.8.26", "hasInstallScript": true, "license": "MIT", "workspaces": [ @@ -26657,7 +26657,7 @@ }, "open-sse": { "name": "@omniroute/open-sse", - "version": "3.8.25" + "version": "3.8.26" } } } diff --git a/package.json b/package.json index 617e5f72ca1..a5f1cdecca0 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "omniroute", - "version": "3.8.25", + "version": "3.8.26", "description": "Unified AI router with 160+ providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { diff --git a/scripts/check/check-cognitive-complexity.mjs b/scripts/check/check-cognitive-complexity.mjs index 90ea5d3e551..663fd8eb441 100644 --- a/scripts/check/check-cognitive-complexity.mjs +++ b/scripts/check/check-cognitive-complexity.mjs @@ -31,7 +31,7 @@ const ESLINT_BIN = path.join(ROOT, "node_modules", ".bin", "eslint"); const BASELINE_PATH = path.resolve( process.argv.includes("--baseline") ? process.argv[process.argv.indexOf("--baseline") + 1] - : path.join(ROOT, "quality-baseline.json") + : path.join(ROOT, "config/quality/quality-baseline.json") ); const ESLINT_ARGS = [ diff --git a/scripts/check/check-complexity.mjs b/scripts/check/check-complexity.mjs index bc2f8c42b47..375512bdf84 100644 --- a/scripts/check/check-complexity.mjs +++ b/scripts/check/check-complexity.mjs @@ -19,7 +19,7 @@ const ROOT = process.cwd(); const BASELINE_PATH = path.resolve( process.argv.includes("--baseline") ? process.argv[process.argv.indexOf("--baseline") + 1] - : path.join(ROOT, "complexity-baseline.json") + : path.join(ROOT, "config/quality/complexity-baseline.json") ); const UPDATE = process.argv.includes("--update"); const CONFIG_PATH = path.join(ROOT, "eslint.complexity.config.mjs"); diff --git a/scripts/check/check-dead-code.mjs b/scripts/check/check-dead-code.mjs index 3e93890a07f..1c7f5c5e34b 100644 --- a/scripts/check/check-dead-code.mjs +++ b/scripts/check/check-dead-code.mjs @@ -28,7 +28,7 @@ const UPDATE = process.argv.includes("--update"); const BASELINE_PATH = path.resolve( process.argv.includes("--baseline") ? process.argv[process.argv.indexOf("--baseline") + 1] - : path.join(ROOT, "quality-baseline.json") + : path.join(ROOT, "config/quality/quality-baseline.json") ); /** diff --git a/scripts/check/check-deps.mjs b/scripts/check/check-deps.mjs index 1cc88f29c87..cb3eb280964 100644 --- a/scripts/check/check-deps.mjs +++ b/scripts/check/check-deps.mjs @@ -29,7 +29,7 @@ import { execFileSync } from "node:child_process"; import { assertNoStale } from "./lib/allowlist.mjs"; const ROOT = process.cwd(); -const ALLOWLIST_PATH = path.join(ROOT, "dependency-allowlist.json"); +const ALLOWLIST_PATH = path.join(ROOT, "config/quality/dependency-allowlist.json"); // Directories to exclude when discovering package.json files. // Using a set of path segment prefixes (relative to ROOT, forward slashes). @@ -141,16 +141,12 @@ export function queryNpmRegistry(pkgName, timeoutMs = 8000) { // Scope packages need URL-encoding for the `npm view` command. // `npm view` accepts scoped packages natively — no encoding needed. try { - const raw = execFileSync( - "npm", - ["view", pkgName, "time.created", "--json"], - { - encoding: "utf8", - timeout: timeoutMs, - // Suppress npm progress/warn output on stderr - stdio: ["ignore", "pipe", "pipe"], - } - ); + const raw = execFileSync("npm", ["view", pkgName, "time.created", "--json"], { + encoding: "utf8", + timeout: timeoutMs, + // Suppress npm progress/warn output on stderr + stdio: ["ignore", "pipe", "pipe"], + }); // npm view --json emits a quoted string or null/empty for missing fields const trimmed = raw.trim(); if (!trimmed) { @@ -165,7 +161,11 @@ export function queryNpmRegistry(pkgName, timeoutMs = 8000) { // npm exits with code 1 when the package is NOT found ("E404") const stderr = err.stderr?.toString() || ""; const stdout = err.stdout?.toString() || ""; - if (stderr.includes("E404") || stdout.includes("E404") || stderr.includes("npm ERR! code E404")) { + if ( + stderr.includes("E404") || + stdout.includes("E404") || + stderr.includes("npm ERR! code E404") + ) { return { exists: false, createdMs: null }; } // Any other error (ETIMEDOUT, ENOTFOUND, etc.) = network/offline — return null diff --git a/scripts/check/check-duplication.mjs b/scripts/check/check-duplication.mjs index 1dc283a59cd..60b5a7bf5ce 100644 --- a/scripts/check/check-duplication.mjs +++ b/scripts/check/check-duplication.mjs @@ -15,13 +15,23 @@ const ROOT = process.cwd(); const BASELINE_PATH = path.resolve( process.argv.includes("--baseline") ? process.argv[process.argv.indexOf("--baseline") + 1] - : path.join(ROOT, "duplication-baseline.json") + : path.join(ROOT, "config/quality/duplication-baseline.json") ); const UPDATE = process.argv.includes("--update"); const EPS = 0.05; // tolerância de ruído de float (jscpd é determinístico; isto é margem) // Use local binary (pinned in package.json devDependencies — no registry download at CI time) const JSCPD_BIN = path.join(ROOT, "node_modules", ".bin", "jscpd"); -const JSCPD_FIXED_ARGS = ["src", "open-sse", "--reporters", "json", "--silent", "--min-tokens", "50", "--ignore", "**/*.test.ts,**/*.test.tsx,**/__tests__/**"]; +const JSCPD_FIXED_ARGS = [ + "src", + "open-sse", + "--reporters", + "json", + "--silent", + "--min-tokens", + "50", + "--ignore", + "**/*.test.ts,**/*.test.tsx,**/__tests__/**", +]; /** Avalia a % atual contra o baseline. */ export function evaluateDuplication(current, baseline, eps = EPS) { diff --git a/scripts/check/check-fabricated-docs.mjs b/scripts/check/check-fabricated-docs.mjs index 6bac4aa4d9a..cedc6759780 100644 --- a/scripts/check/check-fabricated-docs.mjs +++ b/scripts/check/check-fabricated-docs.mjs @@ -273,8 +273,8 @@ const ENV_VAR_DENYLIST = new Set([ // Gate allowlist constant names (JS identifiers, not env vars) — documented in // docs/architecture/QUALITY_GATES.md and docs/research/DISCOVERY_TOOL_DESIGN.md "KNOWN_STALE_DOC_REFS", // export const in check-docs-symbols.mjs - "KNOWN_MISSING", // export const in check-fetch-targets.mjs - "KNOWN_RAW_SQL", // export const in check-db-rules.mjs + "KNOWN_MISSING", // export const in check-fetch-targets.mjs + "KNOWN_RAW_SQL", // export const in check-db-rules.mjs ]); /** Endpoints that don't follow the standard route.ts pattern. */ @@ -313,6 +313,9 @@ const SKIP_DOC_FILES = new Set([ "docs/reference/PROVIDER_REFERENCE.md", // auto-generated from providers.ts "docs/reference/openapi.yaml", "docs/i18n", // translations — separate workflow + // Point-in-time documentation audit (v3.8.24): intentionally references drift, + // counts, and not-yet-existing files as part of documenting them — not living docs. + "docs/ops/DOCUMENTATION_AUDIT_REPORT.md", ]); // ── File discovery ───────────────────────────────────────────────────────── diff --git a/scripts/check/check-file-size.mjs b/scripts/check/check-file-size.mjs index eabc4c68075..82e4193a2a6 100644 --- a/scripts/check/check-file-size.mjs +++ b/scripts/check/check-file-size.mjs @@ -15,7 +15,9 @@ function getArg(name, fallback) { const i = process.argv.indexOf(name); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback; } -const BASELINE_PATH = path.resolve(getArg("--baseline", path.join(ROOT, "file-size-baseline.json"))); +const BASELINE_PATH = path.resolve( + getArg("--baseline", path.join(ROOT, "config/quality/file-size-baseline.json")) +); const UPDATE = process.argv.includes("--update"); const SCAN_DIRS = ["src", "open-sse", "electron", "bin"]; // Directories to skip when walking — build artifacts and installed packages. @@ -30,7 +32,8 @@ export function evaluateFileSizes(currentLocByFile, frozen, cap) { const improvements = []; for (const [file, loc] of Object.entries(currentLocByFile)) { if (file in frozen) { - if (loc > frozen[file]) violations.push(`${file}: ${loc} > congelado ${frozen[file]} (não pode crescer)`); + if (loc > frozen[file]) + violations.push(`${file}: ${loc} > congelado ${frozen[file]} (não pode crescer)`); else if (loc < frozen[file]) improvements.push([file, loc]); } else if (loc > cap) { violations.push(`${file}: ${loc} > cap ${cap} (arquivo novo acima do limite)`); @@ -49,7 +52,11 @@ function walk(dir, acc = []) { const p = path.join(dir, e.name); if (e.isDirectory()) { if (!SKIP_DIRS.has(e.name)) walk(p, acc); - } else if (/\.(ts|tsx)$/.test(e.name) && !/\.test\.tsx?$/.test(e.name) && !/\.d\.ts$/.test(e.name)) { + } else if ( + /\.(ts|tsx)$/.test(e.name) && + !/\.test\.tsx?$/.test(e.name) && + !/\.d\.ts$/.test(e.name) + ) { acc.push(p); } } @@ -59,7 +66,8 @@ function walk(dir, acc = []) { function collectLoc() { const out = {}; for (const d of SCAN_DIRS) - for (const f of walk(path.join(ROOT, d))) out[path.relative(ROOT, f).replace(/\\/g, "/")] = countLines(f); + for (const f of walk(path.join(ROOT, d))) + out[path.relative(ROOT, f).replace(/\\/g, "/")] = countLines(f); return out; } @@ -76,7 +84,8 @@ function main() { if (UPDATE && violations.length === 0 && improvements.length) { for (const [file, loc] of improvements) { - if (loc <= cap) delete frozen[file]; // caiu para dentro do cap → sai do baseline + if (loc <= cap) + delete frozen[file]; // caiu para dentro do cap → sai do baseline else frozen[file] = loc; // continua grande mas encolheu → trava no novo valor } baseline.frozen = Object.fromEntries(Object.entries(frozen).sort()); diff --git a/scripts/check/check-licenses.mjs b/scripts/check/check-licenses.mjs index 3061314db6c..3c265144bcf 100644 --- a/scripts/check/check-licenses.mjs +++ b/scripts/check/check-licenses.mjs @@ -22,7 +22,7 @@ import path from "node:path"; import { pathToFileURL } from "node:url"; const ROOT = process.cwd(); -const ALLOWLIST_PATH = path.join(ROOT, ".license-allowlist.json"); +const ALLOWLIST_PATH = path.join(ROOT, "config/quality/.license-allowlist.json"); const CHECKER_BIN = path.join(ROOT, "node_modules", ".bin", "license-checker-rseidelsohn"); const VERBOSE = process.argv.includes("--verbose"); @@ -39,7 +39,9 @@ const PRINT_JSON = process.argv.includes("--json"); */ export function loadAllowlist() { if (!fs.existsSync(ALLOWLIST_PATH)) { - throw new Error(`Allowlist not found: ${ALLOWLIST_PATH}. Create .license-allowlist.json first.`); + throw new Error( + `Allowlist not found: ${ALLOWLIST_PATH}. Create .license-allowlist.json first.` + ); } const raw = fs.readFileSync(ALLOWLIST_PATH, "utf-8"); const parsed = JSON.parse(raw); @@ -86,7 +88,10 @@ export function classifyLicense(packageName, license, allowlist) { } // 4. Denied - return { status: "denied", reason: `license '${license}' not in allowlist and no exception registered for '${baseName}'` }; + return { + status: "denied", + reason: `license '${license}' not in allowlist and no exception registered for '${baseName}'`, + }; } /** @@ -195,7 +200,9 @@ function main() { // Print exceptions (informational) if (exceptions.length > 0) { - console.log("\n[check-licenses] Exceções registradas (não bloqueantes, revisar periodicamente):"); + console.log( + "\n[check-licenses] Exceções registradas (não bloqueantes, revisar periodicamente):" + ); for (const { pkgKey, license } of exceptions) { const baseName = stripVersion(pkgKey); const exc = allowlist.exceptions[baseName]; @@ -217,7 +224,9 @@ function main() { // Print violations and fail if (violations.length > 0) { - console.error("\n[check-licenses] ❌ VIOLAÇÕES DE POLÍTICA — deps de produção com licença não permitida:"); + console.error( + "\n[check-licenses] ❌ VIOLAÇÕES DE POLÍTICA — deps de produção com licença não permitida:" + ); for (const { pkgKey, license, reason } of violations) { console.error(` ✗ ${pkgKey}: ${license}`); console.error(` → ${reason}`); @@ -231,11 +240,14 @@ function main() { return; } - console.log("\n[check-licenses] ✅ Todos os pacotes de produção estão em conformidade com a política de licenças."); + console.log( + "\n[check-licenses] ✅ Todos os pacotes de produção estão em conformidade com a política de licenças." + ); } // Run only when invoked directly (not when imported by tests) -const isMain = process.argv[1] === pathToFileURL(import.meta.url).pathname || +const isMain = + process.argv[1] === pathToFileURL(import.meta.url).pathname || process.argv[1]?.endsWith("check-licenses.mjs"); if (isMain) { diff --git a/scripts/check/check-test-discovery.mjs b/scripts/check/check-test-discovery.mjs index 99948e6a2d0..0d0f19d87b1 100644 --- a/scripts/check/check-test-discovery.mjs +++ b/scripts/check/check-test-discovery.mjs @@ -34,7 +34,7 @@ const ROOT = process.cwd(); const BASELINE_PATH = path.resolve( process.argv.includes("--baseline") ? process.argv[process.argv.indexOf("--baseline") + 1] - : path.join(ROOT, "test-discovery-baseline.json") + : path.join(ROOT, "config/quality/test-discovery-baseline.json") ); const UPDATE = process.argv.includes("--update"); diff --git a/scripts/check/check-tracked-artifacts.mjs b/scripts/check/check-tracked-artifacts.mjs index 833ed8dc959..a0324fc168c 100644 --- a/scripts/check/check-tracked-artifacts.mjs +++ b/scripts/check/check-tracked-artifacts.mjs @@ -15,7 +15,10 @@ import { execFileSync } from "node:child_process"; import { pathToFileURL } from "node:url"; const FORBIDDEN_PREFIXES = ["node_modules/", ".next/", "coverage/"]; -const FORBIDDEN_EXACT = new Set(["quality-metrics.json"]); +const FORBIDDEN_EXACT = new Set([ + "quality-metrics.json", // legacy root location (still forbidden if a stale run writes it) + "config/quality/quality-metrics.json", // current generated location (collect-metrics.mjs) +]); /** * Verifica se algum caminho na lista de arquivos rastreados corresponde a um @@ -80,7 +83,9 @@ function main() { process.exit(0); } - console.error(`[tracked-artifacts] FAIL — ${violations.length} forbidden artifact(s) tracked by git:`); + console.error( + `[tracked-artifacts] FAIL — ${violations.length} forbidden artifact(s) tracked by git:` + ); for (const v of violations) { console.error(` ✗ ${v}`); } diff --git a/scripts/check/check-type-coverage.mjs b/scripts/check/check-type-coverage.mjs index f6291c77bad..0051063fd35 100644 --- a/scripts/check/check-type-coverage.mjs +++ b/scripts/check/check-type-coverage.mjs @@ -33,7 +33,7 @@ const UPDATE = process.argv.includes("--update"); const BASELINE_PATH = path.resolve( process.argv.includes("--baseline") ? process.argv[process.argv.indexOf("--baseline") + 1] - : path.join(ROOT, "quality-baseline.json") + : path.join(ROOT, "config/quality/quality-baseline.json") ); // Small epsilon to absorb float noise between runs (type-coverage can vary ~0.01%). diff --git a/scripts/docs/sync-wiki.mjs b/scripts/docs/sync-wiki.mjs new file mode 100644 index 00000000000..ff8f6529dd7 --- /dev/null +++ b/scripts/docs/sync-wiki.mjs @@ -0,0 +1,290 @@ +#!/usr/bin/env node +// scripts/docs/sync-wiki.mjs +// Full GitHub wiki content + cover-count sync. +// +// WHY: the wiki has no generator and historically drifts (it sat at "212+ providers / +// 14 strategies / 37 MCP tools" while code was at 226 / 15 / 87, and new docs like +// SUPPLY_CHAIN never appeared). This closes the loop: content + counts, automated per +// release by .github/workflows/wiki-sync.yml. +// +// DESIGN — update-in-place, never duplicates: +// 1. The wiki page names are hand-curated and NOT deterministically reproducible +// (e.g. "API-Reference" vs "Fly-io-Deployment-Guide"). So we iterate the EXISTING +// wiki pages and fuzzy-match each to a docs/ source by normalized key +// (lowercase, strip non-alphanumeric). When a source exists we rewrite that exact +// page → zero risk of creating a parallel/duplicate page. +// 2. A curated allowlist (NEW_PAGE_EXCLUDE) keeps internal docs (audit reports, plans, +// the docs index) off the public wiki; every other unmatched docs page is ADDED +// with a deterministic acronym-aware name. +// 3. Hand-curated pages with no docs source (Home, _Sidebar, Header, _Footer, +// Languages) are left untouched — except the four cover counts on Home.md. +// 4. EN by default. Localized mirrors (‐Page) are pure update-in-place and +// only touched with --include-i18n (the i18n source lags and is validated +// separately). +// +// Content transform: strip the YAML frontmatter, prepend the wiki language banner. +// +// Usage: +// node scripts/docs/sync-wiki.mjs --wiki-dir # write +// node scripts/docs/sync-wiki.mjs --wiki-dir --dry-run # report only +// node scripts/docs/sync-wiki.mjs --wiki-dir --check # exit 1 on drift +// node scripts/docs/sync-wiki.mjs --wiki-dir --include-i18n # also localized + +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const ROOT = path.resolve(__dirname, "..", ".."); + +// U+2010 HYPHEN separates the locale prefix in localized wiki page names. +const LOCALE_SEP = "‐"; +export const WIKI_BANNER = "> 🌍 [View in other languages](Languages)\n\n\n"; + +// Docs that must never become public wiki pages (internal reports/plans/index). +export const NEW_PAGE_EXCLUDE = new Set([ + "README", // docs index, not a page + "DOCUMENTATION_AUDIT_REPORT", + "DOCUMENTATION_OVERHAUL_PLAN", + "E2E_DASHBOARD_SHAKEDOWN_v3.8.0", + "SUBMIT_PR", + "fix-opencode-context", + "plugins", // docs/dev/plugins.md — internal dev note + "SOCKET_DEV_FINDINGS", +]); + +// Acronyms kept upper-case when minting a NEW page name (existing pages keep their +// curated name via fuzzy match, so this only affects brand-new pages). +const ACRONYMS = new Set([ + "api", "mcp", "a2a", "acp", "cli", "sse", "i18n", "pii", "oauth", "vm", "ai", + "llm", "sdk", "ide", "ui", "ux", "tls", "mitm", "ws", "cors", "jwt", "db", "vps", +]); + +/** Normalized matching key: lowercase, drop extension + every non-alphanumeric char. */ +export function normKey(s) { + return s.toLowerCase().replace(/\.md$/, "").replace(/[^a-z0-9]/g, ""); +} + +/** Deterministic wiki page name for a brand-new page (acronym-aware Title-Case-dashed). */ +export function toWikiName(basename) { + return basename + .replace(/\.md$/, "") + .split(/[_\-\s]+/) + .filter(Boolean) + .map((t) => (ACRONYMS.has(t.toLowerCase()) ? t.toUpperCase() : t[0].toUpperCase() + t.slice(1).toLowerCase())) + .join("-"); +} + +/** Strip YAML frontmatter and prepend the wiki language banner. Pure; exported for tests. */ +export function toWikiContent(docMarkdown) { + const body = docMarkdown.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n/, "").replace(/^\s+/, ""); + return WIKI_BANNER + body.replace(/\s*$/, "") + "\n"; +} + +function read(rel) { + const p = path.join(ROOT, rel); + return fs.existsSync(p) ? fs.readFileSync(p, "utf8") : ""; +} + +// ---- cover-page counts (source of truth) ---- +function providerCount() { + const m = read("docs/reference/PROVIDER_REFERENCE.md").match(/Total providers:\s*\*\*(\d+)\*\*/); + return m ? Number(m[1]) : null; +} +function strategyCount() { + const m = read("src/shared/constants/routingStrategies.ts").match( + /ROUTING_STRATEGY_VALUES\s*=\s*\[([^\]]*)\]/ + ); + return m ? (m[1].match(/"[^"]+"/g) || []).length : null; +} +function localeCount() { + try { + const c = JSON.parse(read("config/i18n.json")); + return Array.isArray(c.locales) ? c.locales.length : null; + } catch { + return null; + } +} +function mcpToolCount() { + // Prefer a literal; the constant is computed at runtime so best-effort only. + const m = read("open-sse/mcp-server/server.ts").match(/TOTAL_MCP_TOOL_COUNT\s*=\s*(\d+)\b/); + return m ? Number(m[1]) : null; +} +export function readCounts() { + return { + providers: providerCount(), + strategies: strategyCount(), + mcpTools: mcpToolCount(), + locales: localeCount(), + }; +} + +/** Apply cover-page count substitutions to Home.md text. Pure; exported for tests. */ +export function syncHomeCounts(home, counts) { + let out = home; + if (counts.providers) { + out = out + .replace(/Connect every AI tool to \d+ providers/g, `Connect every AI tool to ${counts.providers} providers`) + .replace(/\*\*\d+ AI Providers\*\*/g, `**${counts.providers} AI Providers**`) + .replace(/All \d+ supported providers/g, `All ${counts.providers} supported providers`) + .replace(/\b\d+ providers\b/g, `${counts.providers} providers`); + } + if (counts.strategies) { + out = out.replace(/\*\*\d+ Routing Strategies\*\*/g, `**${counts.strategies} Routing Strategies**`); + } + if (counts.mcpTools) { + out = out.replace(/(\|\s*\*\*MCP Server\*\*\s*\|\s*)\d+( tools)/g, `$1${counts.mcpTools}$2`); + } + return out; +} + +// ---- docs discovery ---- +function walkMarkdown(dir, acc = []) { + if (!fs.existsSync(dir)) return acc; + for (const e of fs.readdirSync(dir, { withFileTypes: true })) { + const p = path.join(dir, e.name); + if (e.isDirectory()) walkMarkdown(p, acc); + else if (e.name.endsWith(".md")) acc.push(p); + } + return acc; +} + +/** Build normKey → docs absolute path for English docs (docs/ minus docs/i18n). */ +function indexEnglishDocs() { + const docsRoot = path.join(ROOT, "docs"); + const files = walkMarkdown(docsRoot).filter((f) => !f.includes(`${path.sep}i18n${path.sep}`)); + const byKey = new Map(); + for (const f of files) { + const base = path.basename(f, ".md"); + const k = normKey(base); + // First-writer-wins keeps a deterministic pick for basename collisions. + if (!byKey.has(k)) byKey.set(k, { file: f, base }); + } + return byKey; +} + +/** Build normKey → docs path for a given locale's i18n tree. */ +function indexLocaleDocs(locale) { + const root = path.join(ROOT, "docs", "i18n", locale); + const byKey = new Map(); + for (const f of walkMarkdown(root)) { + const k = normKey(path.basename(f, ".md")); + if (!byKey.has(k)) byKey.set(k, f); + } + return byKey; +} + +function listWikiPages(wikiDir) { + return fs + .readdirSync(wikiDir) + .filter((n) => n.endsWith(".md")) + .map((n) => n.slice(0, -3)); +} + +/** Split a wiki page name into { locale, name }. EN pages have locale = null. */ +export function parseWikiPage(page) { + const idx = page.indexOf(LOCALE_SEP); + if (idx === -1) return { locale: null, name: page }; + return { locale: page.slice(0, idx), name: page.slice(idx + 1) }; +} + +function main() { + const args = process.argv.slice(2); + const wikiDir = args.includes("--wiki-dir") ? args[args.indexOf("--wiki-dir") + 1] : null; + const dryRun = args.includes("--dry-run"); + const check = args.includes("--check"); + const includeI18n = args.includes("--include-i18n"); + const updateExisting = args.includes("--update-existing"); + if (!wikiDir || !fs.existsSync(wikiDir)) { + console.error("usage: sync-wiki.mjs --wiki-dir [--dry-run|--check] [--include-i18n]"); + process.exit(2); + } + + const enDocs = indexEnglishDocs(); + const wikiPages = listWikiPages(wikiDir); + const enWikiKeys = new Set(); + const localeIndexes = new Map(); + + const plan = { update: [], add: [], untouched: [], countsChanged: false }; + + // 1. Update existing wiki pages from their docs source. + for (const page of wikiPages) { + if (page === "Home") continue; // handled by counts below + const { locale, name } = parseWikiPage(page); + const key = normKey(name); + let srcFile = null; + if (!locale) { + enWikiKeys.add(key); + srcFile = enDocs.get(key)?.file ?? null; + } else if (includeI18n) { + if (!localeIndexes.has(locale)) localeIndexes.set(locale, indexLocaleDocs(locale)); + srcFile = localeIndexes.get(locale).get(key) ?? null; + } + if (!srcFile) { + plan.untouched.push(page); + continue; + } + const next = toWikiContent(fs.readFileSync(srcFile, "utf8")); + const cur = fs.readFileSync(path.join(wikiDir, `${page}.md`), "utf8"); + if (next !== cur) plan.update.push({ page, srcFile }); + } + + // 2. Add curated new English pages (unmatched docs, minus the exclude list). + for (const [key, { file, base }] of enDocs) { + if (enWikiKeys.has(key)) continue; + if (NEW_PAGE_EXCLUDE.has(base)) continue; + plan.add.push({ page: toWikiName(base), srcFile: file, base }); + } + + // 3. Home cover counts. + const counts = readCounts(); + const homePath = path.join(wikiDir, "Home.md"); + let homeAfter = null; + if (fs.existsSync(homePath)) { + const before = fs.readFileSync(homePath, "utf8"); + homeAfter = syncHomeCounts(before, counts); + plan.countsChanged = homeAfter !== before; + } + + // ---- report ---- + // Updating existing pages is opt-in: several docs SOURCES carry stale counts (e.g. + // ARCHITECTURE.md still says "177 providers / 37 MCP tools" while the wiki cover was + // hand-patched to 226/87). Overwriting from a staler source would REGRESS the wiki, so + // by default we only ADD missing pages and sync Home counts. Pass --update-existing + // once the docs sources are regenerated (see docs/ops/DOCUMENTATION_AUDIT_REPORT.md). + const updates = updateExisting ? plan.update : []; + const total = updates.length + plan.add.length + (plan.countsChanged ? 1 : 0); + console.log(`[wiki-sync] counts: ${JSON.stringify(counts)}`); + console.log( + `[wiki-sync] add: ${plan.add.length} | Home counts: ${plan.countsChanged ? "drift" : "in-sync"} | ` + + `existing-page updates: ${plan.update.length} (${updateExisting ? "ENABLED" : "skipped — needs --update-existing"}) | untouched: ${plan.untouched.length}` + ); + if (dryRun || check) { + if (plan.add.length) console.log(` add → ${plan.add.map((a) => a.page).join(", ")}`); + if (plan.update.length) + console.log( + ` ${updateExisting ? "update" : "would-update (skipped)"} → ${plan.update.map((u) => u.page).slice(0, 60).join(", ")}${plan.update.length > 60 ? " …" : ""}` + ); + if (check) { + if (total > 0) { + console.error(`✗ wiki out of sync (${total} change(s) pending)`); + process.exit(1); + } + console.log("✓ wiki in sync"); + } + return; + } + + // ---- write ---- + for (const { page, srcFile } of [...updates, ...plan.add]) { + fs.writeFileSync(path.join(wikiDir, `${page}.md`), toWikiContent(fs.readFileSync(srcFile, "utf8"))); + } + if (plan.countsChanged && homeAfter != null) fs.writeFileSync(homePath, homeAfter); + console.log( + `[wiki-sync] wrote ${total} page(s) (add: ${plan.add.length}, updates: ${updates.length}, counts: ${plan.countsChanged ? 1 : 0}).` + ); +} + +const invokedDirectly = + process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); +if (invokedDirectly) main(); diff --git a/scripts/quality/check-quality-ratchet.mjs b/scripts/quality/check-quality-ratchet.mjs index 2e885199078..9948c5bb5a9 100644 --- a/scripts/quality/check-quality-ratchet.mjs +++ b/scripts/quality/check-quality-ratchet.mjs @@ -12,8 +12,12 @@ function getArg(name, fallback) { const i = process.argv.indexOf(name); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback; } -const BASELINE = path.resolve(getArg("--baseline", path.join(cwd, "quality-baseline.json"))); -const METRICS = path.resolve(getArg("--metrics", path.join(cwd, "quality-metrics.json"))); +const BASELINE = path.resolve( + getArg("--baseline", path.join(cwd, "config/quality/quality-baseline.json")) +); +const METRICS = path.resolve( + getArg("--metrics", path.join(cwd, "config/quality/quality-metrics.json")) +); const SUMMARY = getArg("--summary", null); const UPDATE = process.argv.includes("--update"); // --allow-missing: pula métricas do baseline ausentes do metrics (em vez de falhar). @@ -73,7 +77,7 @@ for (const [key, spec] of Object.entries(baseline.metrics)) { status = "↑ melhorou"; if (REQUIRE_TIGHTEN && base - current > tightenSlack) { tightenFailures.push( - `${key}: melhorou de ${base} para ${current} (delta ${(base - current).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR`, + `${key}: melhorou de ${base} para ${current} (delta ${(base - current).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR` ); } } @@ -86,7 +90,7 @@ for (const [key, spec] of Object.entries(baseline.metrics)) { status = "↑ melhorou"; if (REQUIRE_TIGHTEN && current - base > tightenSlack) { tightenFailures.push( - `${key}: melhorou de ${base} para ${current} (delta ${(current - base).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR`, + `${key}: melhorou de ${base} para ${current} (delta ${(current - base).toFixed(4)} > slack ${tightenSlack}) — rode 'npm run quality:ratchet -- --update' e commite o baseline apertado neste PR` ); } } @@ -99,10 +103,10 @@ const baselineKeys = new Set(Object.keys(baseline.metrics)); const orphans = Object.keys(metrics).filter((k) => !baselineKeys.has(k)); if (orphans.length > 0) { console.warn( - `[quality-ratchet] WARN: ${orphans.length} métrica(s) órfã(s) — presente(s) em ${path.basename(METRICS)} mas sem entrada no baseline: ${orphans.join(", ")}`, + `[quality-ratchet] WARN: ${orphans.length} métrica(s) órfã(s) — presente(s) em ${path.basename(METRICS)} mas sem entrada no baseline: ${orphans.join(", ")}` ); console.warn( - `[quality-ratchet] WARN: adicione ${orphans.length === 1 ? "essa métrica" : "essas métricas"} ao baseline (com value/direction) para que sejam catraceadas.`, + `[quality-ratchet] WARN: adicione ${orphans.length === 1 ? "essa métrica" : "essas métricas"} ao baseline (com value/direction) para que sejam catraceadas.` ); } @@ -140,7 +144,7 @@ if (failures.length) { if (REQUIRE_TIGHTEN && !UPDATE && tightenFailures.length > 0) { console.error( "[quality-ratchet] FALHOU (--require-tighten): métrica(s) melhoraram mas o baseline não foi apertado:\n" + - tightenFailures.map((f) => " ✗ " + f).join("\n"), + tightenFailures.map((f) => " ✗ " + f).join("\n") ); process.exit(1); } diff --git a/scripts/quality/collect-metrics.mjs b/scripts/quality/collect-metrics.mjs index bf24f96b29b..2807409831e 100644 --- a/scripts/quality/collect-metrics.mjs +++ b/scripts/quality/collect-metrics.mjs @@ -250,6 +250,9 @@ if (import.meta.url === pathToFileURL(process.argv[1] || "").href) { coverageByModule(); openapiCoverage(); await i18nUiCoverage(); - fs.writeFileSync(path.join(cwd, "quality-metrics.json"), JSON.stringify(out, null, 2) + "\n"); + fs.writeFileSync( + path.join(cwd, "config/quality/quality-metrics.json"), + JSON.stringify(out, null, 2) + "\n" + ); console.log("[collect-metrics]", JSON.stringify(out)); } diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/OpenRouterPresetInput.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/OpenRouterPresetInput.tsx new file mode 100644 index 00000000000..280f420f690 --- /dev/null +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/OpenRouterPresetInput.tsx @@ -0,0 +1,54 @@ +"use client"; + +import { useCallback, useMemo, useState } from "react"; +import { Input } from "@/shared/components"; +import { OPENROUTER_PRESET_MAX_LENGTH } from "@/shared/constants/openRouterPreset"; +import { providerText, type ProviderMessageTranslator } from "../providerPageHelpers"; + +interface OpenRouterPresetInputProps { + value: string; + onChange: (value: string) => void; + t: ProviderMessageTranslator; +} + +export default function OpenRouterPresetInput({ value, onChange, t }: OpenRouterPresetInputProps) { + return ( + onChange(e.target.value)} + placeholder="email-copywriter" + maxLength={OPENROUTER_PRESET_MAX_LENGTH} + hint={providerText( + t, + "openRouterPresetHint", + "Sends this connection's preset as the OpenRouter top-level preset field." + )} + /> + ); +} + +export function useOpenRouterPresetControl( + provider: string | null | undefined, + t: ProviderMessageTranslator +) { + const [value, setValue] = useState(""); + const isOpenRouter = provider === "openrouter"; + const applyTo = useCallback( + (data: Record) => { + const preset = value.trim(); + if (isOpenRouter && preset) data.preset = preset; + }, + [isOpenRouter, value] + ); + const getPatch = useCallback(() => { + if (!isOpenRouter) return {}; + const preset = value.trim(); + return { preset: preset || null }; + }, [isOpenRouter, value]); + const input = useMemo( + () => (isOpenRouter ? : null), + [isOpenRouter, t, value] + ); + return { applyTo, getPatch, input, isOpenRouter, setValue }; +} diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx index df432143dcd..59d94f0131f 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx @@ -1,15 +1,9 @@ "use client"; -// Issue #3501 Phase 1c — extracted from the god-component. -// ~787-LOC modal for adding a new API key / credential to a provider. - import { useState, useEffect, useRef } from "react"; import { useTranslations } from "next-intl"; import { Button, Badge, Input, Modal, Toggle } from "@/shared/components"; -import { - providerAllowsOptionalApiKey, - supportsBulkApiKey, -} from "@/shared/constants/providers"; +import { providerAllowsOptionalApiKey, supportsBulkApiKey } from "@/shared/constants/providers"; import { parseBulkApiKeys } from "@/shared/utils/bulkApiKeyParser"; import { isBaseUrlConfigurableProvider, @@ -30,6 +24,7 @@ import { type CommandCodeAuthFlowState, } from "../../providerPageHelpers"; import { getWebSessionCredentialRequirement } from "../../webSessionCredentials"; +import { useOpenRouterPresetControl } from "../OpenRouterPresetInput"; import WebSessionCredentialGuide from "../WebSessionCredentialGuide"; export interface AddApiKeyModalProps { @@ -76,6 +71,7 @@ export default function AddApiKeyModal({ const defaultRegion = isBedrock ? "eu-west-2" : "us-central1"; const isGlm = isGlmProvider(provider); const isQoder = provider === "qoder"; + const openRouterPreset = useOpenRouterPresetControl(provider, t); const isCloudflare = provider === "cloudflare-ai"; const localProviderMetadata = getLocalProviderMetadata(provider); const isLocalSelfHostedProvider = !!localProviderMetadata; @@ -133,7 +129,6 @@ export default function AddApiKeyModal({ baseUrl: initialBaseUrl || defaultBaseUrl, })); }, [defaultBaseUrl, initialBaseUrl, isOpen]); - const bulkSupported = supportsBulkApiKey(provider); const [mode, setMode] = useState<"single" | "bulk">("single"); const [bulkText, setBulkText] = useState(""); @@ -278,7 +273,6 @@ export default function AddApiKeyModal({ if (!isValid) { if (apiKeyOptional && !credentialInput) { - // Bypass validation block for local/optional providers when no key is provided console.debug("Validation failed but apiKey is optional; proceeding to save."); } else { setSaveError(validationError || credentialValidationFailedMessage); @@ -290,6 +284,7 @@ export default function AddApiKeyModal({ if (formData.customUserAgent.trim()) { providerSpecificData.customUserAgent = formData.customUserAgent.trim(); } + openRouterPreset.applyTo(providerSpecificData); if (formData.routingTags.trim()) { providerSpecificData.tags = parseRoutingTagsInput(formData.routingTags); } @@ -347,15 +342,18 @@ export default function AddApiKeyModal({ setSaveError(null); try { - let providerSpecificData: Record | undefined; + const bulkProviderSpecificData: Record = {}; if (usesBaseUrl) { const checked = normalizeAndValidateHttpBaseUrl(formData.baseUrl, defaultBaseUrl); if (checked.error) { setSaveError(checked.error); return; } - providerSpecificData = { baseUrl: checked.value }; + bulkProviderSpecificData.baseUrl = checked.value; } + openRouterPreset.applyTo(bulkProviderSpecificData); + const providerSpecificData = + Object.keys(bulkProviderSpecificData).length > 0 ? bulkProviderSpecificData : undefined; const res = await fetch("/api/providers/bulk", { method: "POST", @@ -432,6 +430,7 @@ export default function AddApiKeyModal({ {bulkSupported && mode === "bulk" && (

{t("bulkAddFormatHint")}

+ {openRouterPreset.input}