diff --git a/AGENTS.md b/AGENTS.md
index f10fb39483c..d468f7d1984 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -46,7 +46,7 @@ Repository map and Reference Documentation sections below.
## Project at a Glance
-**OmniRoute** — unified AI proxy/router. One endpoint, 348 LLM providers, auto-fallback.
+**OmniRoute** — unified AI proxy/router. One endpoint, 349 LLM providers, auto-fallback.
| Layer | Location | Purpose |
| ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
diff --git a/PROVIDER_REFERENCE.md b/PROVIDER_REFERENCE.md
new file mode 100644
index 00000000000..571fe0e904c
--- /dev/null
+++ b/PROVIDER_REFERENCE.md
@@ -0,0 +1,447 @@
+---
+title: "Provider Reference"
+version: 3.8.50
+lastUpdated: 2026-08-21
+---
+
+# Provider Reference
+
+> **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand.
+> Regenerate with: `npm run gen:provider-reference`
+> **Last generated:** 2026-08-21
+
+Total providers: **349**. See category breakdown below.
+
+## Categories
+
+- **Free** — free tier with API key (configured via dashboard)
+- **No-auth** — public endpoints that require no key or sign-in at all
+- **OAuth** — sign-in flow handled by OmniRoute, no API key needed
+- **Web cookie** — wraps the provider's web app via cookie auth
+- **API key** — paid provider configured via API key (free credits may apply)
+- **Local** — runs on the user's machine (Ollama, LM Studio, vLLM, etc.)
+- **Search** — web search providers
+- **Audio** — audio-only providers (TTS/STT)
+- **Upstream proxy** — providers that proxy to other providers
+- **Cloud agent** — long-running coding agents (Codex Cloud, Devin, Jules)
+- **System** — OmniRoute-internal providers (loopback, etc.)
+
+Additional tags: `image`, `video`, `aggregator`, `enterprise`, `embed/rerank`, `self-hosted`.
+
+`Tool calling` (where shown): `native` — real function-calling API; `emulated` — the `tools` array is prompt-emulated via `webTools.ts` (regex-parsed `{...}` blocks); `none` — `tools` is currently silently dropped. See #7286.
+
+Use the dashboard at `/dashboard/providers` to enable, configure, and test each provider.
+
+---
+
+## No-auth Providers (no key required) (11)
+
+| ID | Alias | Name | Tags | Website | Notes | Tool calling |
+|----|-------|------|------|---------|-------|--------------|
+| `aihorde` | `horde` | AI Horde | No-auth | [link](https://aihorde.net) | No API key required — uses AI Horde's documented anonymous key. Adding a free aihorde.net key is optional and only buys higher queue priority (kudos). | — |
+| `auggie` | `aug` | Augment (Auggie CLI) | No-auth | [link](https://augmentcode.com) | No API key stored by OmniRoute. Install the Auggie CLI and run `auggie login` on this machine, then OmniRoute spawns it locally for each request. | — |
+| `chipotle` | `pepper` | Chipotle Pepper AI (Free) | No-auth | [link](https://amelia.chipotle.com) | No credentials required. Uses Chipotle's public support chatbot via reverse-engineered SockJS/STOMP protocol. | — |
+| `cloudflare-playground` | `cfp` | Cloudflare AI Playground | No-auth | [link](https://playground.ai.cloudflare.com) | No credentials required — anonymous browser sessions over a reverse-engineered cf_agent WebSocket protocol (Playwright transport). | — |
+| `devin-cli-agentic` | `dva` | Devin CLI Agentic Bridge | No-auth | [link](https://docs.devin.ai/work-with-devin/devin-cli) | Authentication is owned by the official Devin CLI in its isolated bridge volume. | emulated |
+| `duckduckgo-web` | `ddgw` | DuckDuckGo AI Chat | No-auth | [link](https://duckduckgo.com/duckchat) | No credentials required — DuckDuckGo AI Chat is anonymous and free. | emulated |
+| `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — |
+| `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — |
+| `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — |
+| `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — |
+| `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — |
+
+## OAuth Providers (25)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). |
+| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. |
+| `antigravity` | — | Antigravity | OAuth | — | — |
+| `claude` | `cc` | Claude Code | OAuth | — | — |
+| `cline` | `cl` | Cline | OAuth | — | — |
+| `clinepass` | `cp` | ClinePass | OAuth | [link](https://cline.bot/cline-pass) | ClinePass is Cline's $9.99/mo subscription bundling 10 open coding models. Sign in with your Cline account (same login as the Cline CLI/IDE), or paste a direct ClinePass API key (app.cline.bot → Settings → API Keys). A ClinePass subscription unlocks the cline-pass/* models. Reuses the Cline WorkOS OAuth flow. |
+| `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. |
+| `codex` | `cx` | OpenAI Codex | OAuth | — | — |
+| `cursor` | `cu` | Cursor IDE | OAuth | — | — |
+| `devin-cli` | `dv` | Devin CLI | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai |
+| `devin-desktop` | — | Devin Desktop | OAuth | [link](https://devin.ai) | Paste an existing Devin API key from an authenticated Devin session. Key export availability and steps vary by Devin version and account. |
+| `ghe-copilot` | `ghe-copilot` | GitHub Enterprise Copilot | OAuth | — | Enter your GHE instance URL (e.g., https://ghe.company.com) in provider settings, then authenticate via device flow. |
+| `github` | `gh` | GitHub Copilot | OAuth | — | — |
+| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab Duo OAuth is not configured. Register an OAuth application at https://gitlab.com/-/profile/applications with redirect URI http://localhost:20128/callback and scopes "ai_features read_user", then set GITLAB_DUO_OAUTH_CLIENT_ID (and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET) and restart. |
+| `grok-cli` | `gc` | Grok Build | OAuth | — | Sign in with your browser, or paste your ~/.grok/auth.json (or the JWT access token) from the Grok Build CLI; refresh_token is rotated automatically either way. |
+| `kilocode` | `kc` | Kilo Code | OAuth | — | — |
+| `kimi-coding` | `kmc` | Kimi Code CLI | OAuth | [link](https://www.kimi.com/code?aff=omniroute) | Sign in with the same Kimi account used by Kimi Code CLI. OmniRoute uses the CLI OAuth flow and Kimi Coding Plan endpoints. |
+| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. |
+| `openference` | `of` | Openference | OAuth | [link](https://openference.com) | Sign in with your Openference account to route requests through api.openference.com. An active plan is required for inference — OAuth may authenticate but return 402 without one. |
+| `qoder` | `if` | Qoder | OAuth | — | — |
+| `raycast` | `rc` | Raycast Pro AI | OAuth | [link](https://raycast.com/ai) | Unofficial integration — uses your Raycast Pro subscription via credentials from the macOS app (Auto-Import or manual capture). May break on Raycast updates. Not for redistribution; personal use only. |
+| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. |
+| `xai-oauth` | `xao` | xAI OAuth (Grok) | OAuth | [link](https://x.ai) | Sign in with xAI to use api.x.ai models such as Grok 4.5. This is separate from Grok Build JWT sessions, which use cli-chat-proxy.grok.com and grok-build model aliases. |
+| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. |
+| `zed-hosted` | — | Zed Hosted Models | OAuth | [link](https://zed.dev) | Sign in with your Zed account (native-app sign-in). OmniRoute generates a one-time RSA keypair and opens zed.dev to authorize it — on a remote/headless install, copy the resulting 127.0.0.1 callback URL from your browser's address bar and paste it back here. Distinct from the 'Zed IDE' credential-import entry above: this proxies chat completions through Zed's own hosted model aggregator (cloud.zed.dev), fronting Anthropic/OpenAI/Google/xAI models under your Zed plan. |
+
+## Web Cookie Providers (35)
+
+| ID | Alias | Name | Tags | Website | Notes | Tool calling |
+|----|-------|------|------|---------|-------|--------------|
+| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | emulated |
+| `adobe-firefly` | `firefly` | Adobe Firefly (Image/Video) | Web cookie | [link](https://firefly.adobe.com) | RECOMMENDED: firefly.adobe.com signed-in → F12 → Network → click firefly-3p.ff.adobe.io (generate-async or models/discovery) → Request Headers → Authorization → copy the token AFTER 'Bearer ' (starts with eyJ…). Cookie-only from firefly.adobe.com mints a GUEST token → 401/403; only multi-domain IMS cookies (adobelogin.com) or that Bearer JWT work. Unofficial/experimental media + Limits. | — |
+| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | emulated |
+| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | emulated |
+| `chatgpt-web-codex` | `cgpt-codex` | ChatGPT Web (Codex) | Web cookie | [link](https://chatgpt.com) | Paste the full ChatGPT Cookie header. OmniRoute verifies it in an isolated headless browser profile. | native |
+| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | none |
+| `conol-web` | `cnl` | Conol (Unofficial/Experimental) | Web cookie | [link](https://conol.ai) | Use browser sign-in, or paste the full Cookie header from conol.ai. The __Secure-better-auth.session_token cookie is required. | — |
+| `copilot-m365-web` | `m365copilot` | Microsoft 365 Copilot (BizChat) | Web cookie | [link](https://m365.cloud.microsoft/chat) | Sign in at m365.cloud.microsoft/chat, then open DevTools → Network → filter 'WS' → click the Chathub WebSocket connection. Copy both the access_token query parameter AND the account-specific Chathub path segment from its request URL (wss://…/Chathub/?…&access_token=…). It is NOT an Authorization: Bearer header on an XHR/Fetch request. The token is short-lived; this is an unofficial integration. Optional: store a refresh_token in providerSpecificData.refreshToken (any Microsoft device-code/refresh flow for the substrate.office.com/sydney scopes) and OmniRoute pre-flight-refreshes the access token itself — otherwise re-capture after every ~75 min expiry. | — |
+| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste the access_token from an authenticated copilot.microsoft.com request (DevTools → Network → Authorization), or export a HAR while logged in | — |
+| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | emulated |
+| `doubao-web` | `db` | Dola Web (ByteDance) | Web cookie | [link](https://www.dola.com) | Paste the full Cookie header from www.dola.com. It should include sessionid, ttwid, and s_v_web_id. If s_v_web_id is unavailable, fp=verify_... from a chat/completion request URL can be used as a fallback. | — |
+| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — |
+| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated |
+| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — |
+| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — |
+| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — |
+| `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — |
+| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated |
+| `kimi-web` | `kimi-web` | Kimi Web | Web cookie | [link](https://www.kimi.com/code?aff=omniroute) | Paste access_token from www.kimi.com DevTools → Application → Local Storage. A legacy kimi-auth cookie is also accepted. | — |
+| `lmarena` | `lma` | Arena (Free) | Web cookie | [link](https://arena.ai) | Paste the full Cookie header from arena.ai (DevTools → Network → request → Cookie). Include arena-auth-prod-v1.0/.1… and cf_clearance/__cf_bm when present. OmniRoute uses Chrome TLS impersonation; if Arena still 403s, set providerSpecificData.recaptchaV3Token from a live browser session. | — |
+| `microsoft-designer-web` | `msdesigner` | Microsoft Designer (Image Generation) | Web cookie | [link](https://designer.microsoft.com) | Sign in at designer.microsoft.com, then open DevTools → Network, generate an image, and find the request to DallE.ashx?action=GetDallEImagesCogSci. Copy the value of its Authorization: Bearer header (the access_token — no 'Bearer ' prefix). The token is short-lived; this is an unofficial, reverse-engineered integration. | — |
+| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your ecto_1_sess cookie AND the ecto1:... WS auth token from meta.ai. Capture the ecto1: token in DevTools → Network → WS → the clippy request's Authorization query param. Example: ecto_1_sess=4240a308...NVDg0; ecto1:ABCD... | emulated |
+| `notion-web` | `nw` | Notion AI Web (Unofficial/Experimental) | Web cookie | [link](https://www.notion.so) | Paste only the token_v2 cookie VALUE from app.notion.com (DevTools → Application → Cookies → token_v2). Do not paste token_v2= or the full Cookie header. Workspace is auto-detected; space_id / notion_user_id are optional. | — |
+| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | emulated |
+| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | — |
+| `promptql` | `pql` | PromptQL (Unofficial/Experimental) | Web cookie | [link](https://prompt.ql.app) | Paste the Bearer JWT from prompt.ql.app DevTools → Network → graphql → Authorization (token only). Optional projectId + session Cookie for refresh. | — |
+| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | emulated |
+| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | emulated |
+| `tencent-aistudio-web` | `tasw` | Tencent AI Studio (Free) | Web cookie | [link](https://aistudio.tencent.ai) | Log in to aistudio.tencent.ai, open DevTools -> Network, copy any request Cookie header containing session tokens. | — |
+| `tinycms-web` | `tcw` | TinyCMS Web (Free/Sub) | Web cookie | [link](https://site.tinycms.xyz) | Go to site.tinycms.xyz, open DevTools → Application → Local Storage, copy the value of 'app-config-uuid' (starts with 'R'), and paste it here. | — |
+| `v0-vercel-web` | `v0-vercel-web` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | — |
+| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | — |
+| `yuanbao-web` | `ybw` | Tencent Yuanbao (Free) | Web cookie | [link](https://yuanbao.tencent.com) | Log in to yuanbao.tencent.com, then paste the full Cookie header (DevTools → Network → any /api request → Request Headers → Cookie). It must contain hy_user and hy_token. | — |
+| `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — |
+| `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — |
+
+## API Key Providers (paid / paid-with-free-credits) (233)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn |
+| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway |
+| `agnes` | `agnes` | Agnes AI | API key, video | [link](https://agnes-ai.com) | Get API key at agnes-ai.com |
+| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required |
+| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | Free tier paused (2026) — AI/ML API is now pay-as-you-go only (min $20 top-up); no recurring free credits. |
+| `ainative` | `ainative` | AINative Studio | API key | [link](https://ainative.studio) | Create a free API key at ainative.studio (no card), then paste it here as a Bearer token. |
+| `aion` | `aion` | Aion Labs | API key | [link](https://www.aionlabs.ai) | Create a free API key at aionlabs.ai (no card), then paste it here as a Bearer token. |
+| `alibaba` | `ali` | Alibaba Cloud Model Studio | API key | [link](https://bailian.console.alibabacloud.com/) | — |
+| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.console.aliyun.com/) | — |
+| `ant-ling` | `ling` | Ant Ling / Ring (inclusionAI) | API key | [link](https://developer.ant-ling.com/en/docs/) | Register and create an API key at the Ant Ling API console (https://chat.ant-ling.com/open), then paste it here. OmniRoute routes chat traffic to https://api.ant-ling.com/v1/chat/completions; the provider is OpenAI-compatible and also exposes an Anthropic-compatible surface. |
+| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — |
+| `anyapi` | `anyapi` | AnyAPI AI | API key, aggregator | [link](https://anyapi.ai) | Free plan: 100,000 ANY Tokens/day and 100 RPM for eligible Free/Basic models; no credit card required. |
+| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 |
+| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai |
+| `auriko` | `auriko` | Auriko | API key, aggregator | [link](https://www.auriko.ai) | Free plan publishes 1,000 Platform RPM and 10,000 BYOK RPM. Platform inference still passes through provider cost; this is not a free-token pool or unlimited free inference. |
+| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. |
+| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. |
+| `bai` | `bai` | b.ai | API key | [link](https://b.ai) | Bearer API key for the b.ai OpenAI-compatible LLM gateway (distinct from TheB.AI). Create a key at https://docs.b.ai, then use https://api.b.ai/v1 as the OpenAI-compatible base URL. |
+| `baichuan` | `baichuan` | Baichuan | API key | [link](https://www.baichuan-ai.com/) | Get API key at platform.baichuan-ai.com |
+| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://ernie.baidu.com/) | Get API key at console.bce.baidu.com |
+| `bailian-coding-plan` | `bcp` | Alibaba Token Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/token-plan-overview) | — |
+| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference |
+| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. |
+| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. |
+| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — |
+| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply |
+| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 |
+| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — |
+| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks |
+| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. |
+| `charm-hyper` | `charm-hyper` | Charm Hyper | API key | [link](https://hyper.charm.land) | 100 free monthly Hypercredits on signup |
+| `chat-oripe` | `chat-oripe` | Chat Oripe | API key, aggregator | [link](https://api.oriper.com) | Official metadata advertises 2M tokens/month, but the public site and documentation were blocked during audit; treat the quota and brand mapping as unconfirmed. |
+| `chatanywhere` | `chatanywhere` | ChatAnywhere | API key, aggregator | [link](https://chatanywhere.tech) | Personal, educational or research use only: public documentation cites 10,000 points/day and 200 requests/day per IP/key; do not use for commercial traffic. |
+| `cheaperinference` | `cinf` | Cheaper Inference | API key | [link](https://cheaperinference.com/?utm_source=omniroute) | — |
+| `chenzk` | `chenzk` | Chenzk API | API key | [link](https://chenzk.top) | — |
+| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. |
+| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . |
+| `cloudcode-one` | `cloudcode-one` | CloudCode.ONE | API key, aggregator | [link](https://cloudcode.one) | Published free models include glm-4.7-flash and glm-4.6v-flash; no numeric quota is published, and key creation may require credit or a coupon. |
+| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) |
+| `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — |
+| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — |
+| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required |
+| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. |
+| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api |
+| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — |
+| `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. |
+| `dahl` | `dahl` | Dahl | API key | [link](https://inference.dahl.global) | Click 'Add Account' to auto-generate a token, or add a manual API key. |
+| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — |
+| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. |
+| `deepai` | `deepai` | DeepAI | API key, image | [link](https://deepai.org) | Use your DeepAI API key. Get one at deepai.org — requires a Pro subscription ($9.99/mo). |
+| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration |
+| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required |
+| `dgrid` | `dgrid` | DGrid | API key | [link](https://dgrid.ai) | DGrid Free Models Router: 10 requests/minute and 100 requests/day. A $5 lifetime top-up unlocks up to 20 requests/minute and 1,000 requests/day. |
+| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. |
+| `digitalocean` | `digitalocean` | DigitalOcean | API key | [link](https://docs.digitalocean.com/products/ai-platform/) | — |
+| `dit` | `dai` | DIT.ai | API key | [link](https://dit.ai) | Use your dit.ai API key in Authorization: Bearer . Fully OpenAI-compatible — a drop-in replacement, just change the base URL to https://api.dit.ai/v1. |
+| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com |
+| `dxnt` | `dxnt` | DXNT / DX Token | API key, aggregator | [link](https://www.dxnt.com) | Free accounts are documented at 100 calls/day; the quota may increase through invitations and can vary by account. |
+| `electronhub` | `electronhub` | Electron Hub | API key, aggregator | [link](https://www.electronhub.ai) | Free plan: 5 RPM, $0.25 weekly credits and 10 Neutrinos/day for :free models; family budgets also apply. |
+| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. |
+| `factory` | `factory` | Factory | API key | [link](https://factory.ai) | Bearer API key for the Factory OpenAI-compatible gateway. |
+| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — |
+| `fastrouter` | `fastrouter` | FastRouter | API key, aggregator | [link](https://fastrouter.ai) | Models with the :free suffix allow 10 requests/day per organization and model; availability may change. |
+| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required |
+| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. |
+| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing |
+| `free-ai` | `free-ai` | Free.ai | API key, aggregator | [link](https://free.ai) | 30,000 tokens/day cover self-hosted models after email verification. Usage beyond the pool can bill at raw cost, and premium external models are paid. |
+| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — |
+| `freebuff` | `freebuff` | Freebuff | API key | [link](https://freebuff.com) | Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester). |
+| `freeinference` | `freeinference` | FreeInference | API key, aggregator | [link](https://freeinference.org) | Free research access without a card; non-Harvard applicants require manual approval and no numeric quota is publicly guaranteed. |
+| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. |
+| `freetheai` | `fta` | FreeTheAi | API key, aggregator | [link](https://freetheai.xyz) | Join the FreeTheAi Discord to get your free API key. |
+| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required |
+| `g4f-gemini` | `g4fgem` | g4f.space — Gemini | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. |
+| `g4f-groq` | `g4fgroq` | g4f.space — Groq | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. |
+| `g4f-nvidia` | `g4fnv` | g4f.space — NVIDIA | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. |
+| `g4f-ollama` | `g4foll` | g4f.space — Ollama | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. |
+| `g4f-pollinations` | `g4fpol` | g4f.space — Pollinations | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. |
+| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | ⚠️ **DEPRECATED.** api.galadriel.ai no longer resolves (sweep 2026-06-19); the inference API appears discontinued. |
+| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free tier available through Google AI Studio; current per-model quotas and regional limits apply |
+| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — |
+| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — |
+| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. |
+| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free MiMo (xiaomi/mimo-v2.5) revoked 2026-05 — Opengateway is now a pay-as-you-go credit gateway; no recurring free model. |
+| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free Nemotron promo ended 2026-06 — the GMI Cloud route is now pay-as-you-go credit only. |
+| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — |
+| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — |
+| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — |
+| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card |
+| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. |
+| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api |
+| `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn |
+| `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. |
+| `helyxai` | `helyxai` | Helyx AI | API key, aggregator | [link](https://helyxai.space) | Operational Free plan documents 100,000 tokens/day; the site's separate 2M+ marketing claim conflicts and is not treated as a quota guarantee. |
+| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — |
+| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) |
+| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference |
+| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api |
+| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn |
+| `inception` | `inception` | Inception | API key | [link](https://docs.inceptionlabs.ai) | 10M free tokens on signup, no credit card required. |
+| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available |
+| `internlm` | `internlm` | InternLM (Intern-S1) | API key | [link](https://internlm.intern-ai.org.cn/) | Free monthly quota ~1M input / 3M output tokens (~10 RPM) |
+| `jina-ai` | `jina` | Jina AI (Foundation API) | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for api.jina.ai — embeddings, rerank, classify, segment, and search. Dashboard keys take precedence over JINA_AI_API_KEY. This is not the Reader / r.jina.ai card and does not fetch URLs. |
+| `jina-reader` | `jr` | Jina Reader (r.jina.ai) | API key | [link](https://jina.ai/reader) | Bearer API key for r.jina.ai URL-to-markdown (/v1/web/fetch only). Does not serve /v1/embeddings or /v1/rerank. The same Jina token as Foundation API works; OmniRoute reuses a jina-ai dashboard key or JINA_AI_API_KEY when this card is empty. |
+| `kenari` | `kenari` | Kenari | API key | [link](https://kenari.id) | Use your Kenari API key (kn-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://kenari.id/v1. |
+| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — |
+| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — |
+| `kimi` | `kimi` | Kimi (Legacy Moonshot API) | API key | [link](https://platform.kimi.ai?aff=omniroute) | — |
+| `kimi-coding-apikey` | `kmca` | Kimi Code API Key | API key | [link](https://www.kimi.com/code?aff=omniroute) | — |
+| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — |
+| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — |
+| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer |
+| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai |
+| `literouter` | `literouter` | LiteRouter | API key, aggregator | [link](https://literouter.com) | Free model variants use the :free suffix; daily credit limits vary by model and free input is capped at 5,000 tokens. |
+| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — |
+| `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. |
+| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. |
+| `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. |
+| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. |
+| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. |
+| `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. |
+| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — |
+| `meganova-ai` | `meganova-ai` | MegaNova AI | API key, aggregator | [link](https://meganova.ai) | Free signup without a card. Published Tier 1 per-model quotas total 550 requests/day; they are not a shared global pool, and paid overage can apply if enabled. |
+| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — |
+| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — |
+| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — |
+| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required |
+| `mixedbread` | `mxbai` | Mixedbread AI | API key | [link](https://www.mixedbread.com) | Bearer API key for the Mixedbread embeddings API. |
+| `mixlayer` | `mixlayer` | Mixlayer | API key, aggregator | [link](https://www.mixlayer.com) | The qwen/qwen3.5-4b-free model is free for prototyping and rate-limited; no fixed public RPM or daily quota is confirmed. |
+| `mnn-ai` | `mnn-ai` | MNN AI | API key, aggregator | [link](https://mnnai.ru) | Free plan: $1 monthly credits, 10 RPM and access only to models marked Free. |
+| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. |
+| `modelscope` | `ms` | ModelScope | API key | [link](https://modelscope.cn) | Free tier via ModelScope API-Inference — Alibaba account required. |
+| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | ⚠️ **DEPRECATED.** Monster API shuttered operations on 2026-06-30. Use alternative OpenAI-compatible providers. |
+| `moonshot` | `moonshot` | Kimi | API key | [link](https://platform.kimi.ai?aff=omniroute) | — |
+| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 |
+| `muse-code` | `mc` | Muse Code (Meta) | API key | [link](https://github.com/meta-llama/llama-stack) | Use your META_API_KEY env var as a Bearer token. Muse Code CLI uses the OpenAI Responses API wire format (POST /responses). |
+| `naga-ac` | `naga` | Naga.ac | API key, aggregator | [link](https://naga.ac) | Get API key at naga.ac — Google/GitHub/Discord signup available. |
+| `naga-ai` | `naga-ai` | Naga AI | API key, aggregator | [link](https://naga.ac) | Models marked :free are publicly listed, but no numeric quota is confirmed. Naga's policy warns that free-tier prompts and outputs may be collected or used for training. |
+| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — |
+| `nara` | `nara` | NaraRouter | API key | [link](https://bynara.id) | Get a free API key via NaraRouter's Telegram channel, then paste it here as a Bearer token. |
+| `navy` | `navy` | NavyAI | API key | [link](https://api.navy) | Create a free API key from the NavyAI dashboard, then paste it here as a Bearer token. |
+| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing |
+| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. |
+| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai |
+| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. |
+| `novita` | `novita` | Novita AI | API key, video, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) |
+| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing |
+| `nube` | `nube` | Nube.sh | API key | [link](https://nube.sh) | — |
+| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) |
+| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. |
+| `ofoxai` | `ofoxai` | OfoxAI | API key, aggregator | [link](https://ofox.ai) | The current catalog advertises 10+ free models without a public numeric quota; review upstream provenance, retention and training terms before production use. |
+| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/keys) | — |
+| `openadapter` | `oad` | OpenAdapter | API key | [link](https://openadapter.dev) | Use your OpenAdapter API key in Authorization: Bearer sk-cv-. Fully OpenAI-compatible. API base URL: https://api.openadapter.in/v1. |
+| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — |
+| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — |
+| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — |
+| `openference-api` | `ofa` | Openference API | API key | [link](https://openference.com) | Free plan: 3-day trial with open-source models — no credit card required |
+| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD |
+| `openvecta` | `openvecta` | OpenVecta | API key | [link](https://openvecta.com) | Free credits on signup for OpenAI-compatible inference across LLMs, embeddings, and reasoning models |
+| `orcarouter` | `orcarouter` | OrcaRouter | API key | [link](https://www.orcarouter.ai) | — |
+| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — |
+| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — |
+| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — |
+| `pioneer` | `pn` | Pioneer AI | API key | [link](https://pioneer.ai) | $75 free usage credits — no credit card required |
+| `plamo` | `plamo` | PLaMo | API key | [link](https://plamo.preferredai.jp/api) | — |
+| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. |
+| `poixe-ai` | `poixe-ai` | Poixe AI | API key, aggregator | [link](https://poixe.com) | Current public free limits are small and model-group specific: 2 RPM/5 RPD for large-cup models and 20 RPM/50 RPD for small-cup models. |
+| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | Anonymous/keyless access to the documented free models is best-effort. Local v3.8.50 verification (2026-07-31) returned 401 via OmniRoute and Cloudflare 1010 on direct upstream probes from the same network. Premium models still require a Pollinations API key from enter.pollinations.ai. |
+| `poolside` | `poolside` | Poolside | API key | [link](https://poolside.ai) | Laguna S 2.1 and XS 2.1 are free during Preview; no public numeric quota is published. |
+| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | ⚠️ **DEPRECATED.** serving.app.predibase.com no longer resolves (sweep 2026-06-19); the managed serving API appears discontinued. |
+| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid |
+| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product-s/qianfan_home) | — |
+| `qiniu` | `qiniu` | Qiniu | API key | [link](https://www.qiniu.com) | — |
+| `qwen-cloud` | `qwc` | Qwen Cloud | API key | [link](https://www.qwencloud.com/) | — |
+| `qwen-cloud-token-plan` | `qct` | Qwen Cloud Token Plan | API key | [link](https://www.qwencloud.com/pricing/token-plan) | — |
+| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — |
+| `regolo` | `regolo` | Regolo AI | API key | [link](https://regolo.ai) | Get your Regolo API key from regolo.ai, then paste it here as a Bearer token. |
+| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. |
+| `requesty` | `requesty` | Requesty | API key | [link](https://requesty.ai) | Free tier ~200 requests/day - multi-model routing gateway (300+ models) |
+| `routeway` | `routeway` | Routeway | API key | [link](https://routeway.ai) | Create a free API key at routeway.ai, then paste it here as a Bearer token. |
+| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. |
+| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required |
+| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. |
+| `sarvam` | `sarvam` | Sarvam AI | API key | [link](https://docs.sarvam.ai) | ₹1,000 in free signup credits — never expire |
+| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/docs/ai-data/generative-apis/) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B |
+| `sealion` | `sealion` | SEA-LION | API key | [link](https://sea-lion.ai) | Sign in at sea-lion.ai with Google (no card, no region wall), create an API key, then paste it here. |
+| `segmind` | `segmind` | Segmind | API key, image, video | [link](https://segmind.com) | Use your Segmind API key in the x-api-key header. OmniRoute targets https://api.segmind.com/v1/ and returns the generated image/video bytes directly. |
+| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn |
+| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus currently listed $0 models after identity verification; availability and limits may change |
+| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — |
+| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn |
+| `speka` | `speka` | Speka AI | API key, aggregator | [link](https://speka.me) | Free plan: $1 monthly usage, 10 RPM, one API key and access to open models and the playground; no card required. |
+| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — |
+| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com |
+| `sumopod` | `sumopod` | SumoPod | API key | [link](https://ai.sumopod.com) | Use your SumoPod API key (sk-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://ai.sumopod.com/v1. |
+| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) |
+| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — |
+| `tabitoken` | `tabitoken` | TabiToken | API key, aggregator | [link](https://tabitoken.com) | — |
+| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com |
+| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. |
+| `tinyfish` | `tf` | TinyFish Fetch | API key | [link](https://docs.tinyfish.ai/fetch-api) | X-API-Key from agent.tinyfish.ai/api-keys |
+| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | — |
+| `token-kiosk` | `tk` | Token Kiosk | API key | [link](https://agent-router.gaib.ai) | Use your Token Kiosk API key in Authorization: Bearer . Fully OpenAI-compatible gateway. API base URL: https://agent-router.gaib.ai/v1. |
+| `tokenreply` | `tokenreply` | TokenReply | API key, aggregator | [link](https://www.tokenreply.com) | Free-tagged models have model- and campaign-specific daily limits; no fixed global free quota is published. |
+| `tokenrouter` | `trk` | TokenRouter | API key | [link](https://tokenrouter.com) | Use your TokenRouter API key in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.tokenrouter.com/v1. |
+| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — |
+| `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. |
+| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) |
+| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. |
+| `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. |
+| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — |
+| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — |
+| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — |
+| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — |
+| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token |
+| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. |
+| `void-ai` | `void-ai` | Void AI | API key, aggregator | [link](https://voidai.app) | The public model catalog marks some models with a free plan requirement, but access is conditional and no numeric quota is confirmed. |
+| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — |
+| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. |
+| `wafer` | `wafer` | Wafer AI | API key | [link](https://wafer.ai) | — |
+| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — |
+| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. |
+| `writer` | `writer` | Writer | API key | [link](https://dev.writer.com) | — |
+| `x5lab` | `x5lab` | X5Lab | API key | [link](https://x5lab.dev) | Use your X5Lab API key (x5-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.x5lab.dev/v1. |
+| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | Use an official xAI API key, or sign in with xAI OAuth. Grok Build JWT sessions remain a separate provider. |
+| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — |
+| `xiaomi-mimo-token-plan` | `mimotp` | Xiaomi MiMo Token Plan | API key | [link](https://mimo.mi.com) | — |
+| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com |
+| `yolo-auto` | `yolo-auto` | Yolo-Auto | API key, aggregator | [link](https://yolo-auto.com) | Free API access is request-limited and intended for testing; no numeric daily quota is published and free access is not promised indefinitely. |
+| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — |
+| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. |
+| `zerolimitai` | `zerolimitai` | ZeroLimitAI | API key, aggregator | [link](https://www.zerolimitai.com) | Temporary free trial is advertised, but official pages conflict between 3 and 7 days; a 100-calls/day claim is not treated as permanent. |
+| `zylo-api` | `zylo` | Zylo API | API key, aggregator | [link](https://zyloai.net) | Basic plan: 10 RPM, 7,200 requests/day and 200,000 tokens/day; limited to Basic text models. |
+
+## Local Providers (14)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). |
+| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). |
+| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). |
+| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. |
+| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). |
+| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). |
+| `mlx-gemma` | `mlx-gemma` | MLX Gemma 26B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11435. Requires uv and mlx-lm installed. Model: mlx-community/gemma-4-26B-A4B-it-qat-q4_0-mlx-aligned (~15.9GB peak memory). |
+| `mlx-qwen` | `mlx-qwen` | MLX Qwen 3.8 27B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11436. Requires uv and mlx-lm installed. Model: maglun/Qwen3.8-27B-MLX-Mixed-3.80bpw (~13.1GB peak memory). |
+| `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. |
+| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). |
+| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). |
+| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). |
+| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). |
+| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). |
+
+## Search Providers (13)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard |
+| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai |
+| `firecrawl` | `fc` | Firecrawl | Search | [link](https://firecrawl.dev) | API key from firecrawl.dev/app/api-keys (or set your self-hosted Firecrawl base URL) |
+| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) |
+| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard |
+| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/keys) | Same API key as Ollama Cloud (from ollama.com/settings/keys) |
+| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) |
+| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs/google) | API key from SearchAPI (query param or Bearer auth) |
+| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. |
+| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard |
+| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) |
+| `x-search` | `x_search` | X Search (Grok) | Search | [link](https://docs.x.ai/developers/tools/x-search) | SuperGrok OAuth (xai-oauth) or xAI API key. This is Grok X Search, not the X Developer MCP. |
+| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/business/api/) | X-API-Key from the You.com platform dashboard |
+
+## Audio-only Providers (12)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — |
+| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. |
+| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — |
+| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — |
+| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — |
+| `fishaudio` | `fishaudio` | Fish Audio | Audio | [link](https://fish.audio) | — |
+| `gladia` | `gladia` | Gladia | Audio | [link](https://gladia.io) | — |
+| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — |
+| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — |
+| `rev-ai` | `revai` | Rev AI | Audio | [link](https://www.rev.ai) | — |
+| `soniox` | `sx` | Soniox | Audio | [link](https://soniox.com) | — |
+| `speechmatics` | `sm` | Speechmatics | Audio | [link](https://www.speechmatics.com) | Free tier — 8 hours/month, no credit card required. Batch (async) mode only. |
+
+## Upstream Proxy Providers (2)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — |
+| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — |
+
+## Cloud Agent Providers (3)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. |
+| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. |
+| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. |
+
+## System Providers (1)
+
+| ID | Alias | Name | Tags | Website | Notes |
+|----|-------|------|------|---------|-------|
+| `auto` | `auto` | Auto (Zero-Config) | System | — | — |
+
+## Sources of truth
+
+- Catalog: [`src/shared/constants/providers.ts`](../../src/shared/constants/providers.ts)
+- Registry (per-model details): [`open-sse/config/providerRegistry.ts`](../../open-sse/config/providerRegistry.ts)
+- Executors: [`open-sse/executors/`](../../open-sse/executors/) (106 implementations)
+- Translators: [`open-sse/translator/`](../../open-sse/translator/)
+
+## See Also
+
+- [FREE_TIERS.md](./FREE_TIERS.md) — curated free-tier guide
+- [USER_GUIDE.md](../guides/USER_GUIDE.md) — provider setup walkthrough
+- [ARCHITECTURE.md](../architecture/ARCHITECTURE.md) — overall architecture
diff --git a/README.md b/README.md
index 55e6d0328a1..6600b406492 100644
--- a/README.md
+++ b/README.md
@@ -7,7 +7,7 @@
# 🚀 OmniRoute — The Free AI Gateway
-
+
@@ -101,7 +101,7 @@
@@ -461,7 +461,7 @@ All **19** strategies — mix & match per combo step:
-
+📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md)
@@ -559,7 +559,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute
- **🖼️ New endpoints** — `/v1/ocr` (Mistral OCR) and `/v1/audio/translations` (Whisper-style) round out the media surface. → [API Reference](docs/reference/API_REFERENCE.md)
- **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md)
- **🌍 Deployment & ops** — reverse-proxy `basePath`, browser-language auto-detect, per-key device tracking, root-less MITM trust, zh-TW localization. → [Environment](docs/reference/ENVIRONMENT.md)
-- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **348-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md)
+- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **349-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md)
- **📡 Routing transparency** — every response carries an `X-OmniRoute-Decision` header naming the strategy/provider/latency that served it, a new `cache-optimized` combo strategy + Auto-Combo `cacheAffinity` factor route repeat requests back to the connection holding the cached prefix, and a read-only `/v1/auto-combo/{channel}/candidates` endpoint exposes an `auto/*` channel's live candidate pool. → [Auto-Combo](docs/routing/AUTO-COMBO.md)
- **⚡ Local performance & infra** — one-click local Redis, Cloudflare Workers / Deno Deploy relay deployers, Bifrost & Mux as supervised embedded services. → [Embedded Services](docs/frameworks/EMBEDDED-SERVICES.md)
@@ -642,11 +642,11 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
-## 🌐 348 AI Providers — 90+ Free
+## 🌐 349 AI Providers — 90+ Free
-> The most complete catalog of any open-source router: **348 providers**, **90+ with a free tier**, **56 free forever**.
+> The most complete catalog of any open-source router: **349 providers**, **90+ with a free tier**, **56 free forever**.
@@ -990,11 +990,11 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \
`:latest` follows the highest **published** stable SemVer. It does not track git `main`. Pin `:X.Y.Z` for GitOps. See [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels).The image pins **`OMNIROUTE_MEMORY_MB=1024`**. That is enough for the dashboard and a light chat. **Coding agents** (`POST /v1/responses` from Claude Code, Codex, Grok, …) need a much larger V8 heap or the process `FATAL ERROR`s at ~12 GiB under two overlapping long contexts. Size the container above the heap (native buffers sit outside V8):
-| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) |
-| --- | --- | --- |
-| Dashboard / light chat | `1024` (image default) | ≥2 g |
-| One coding agent | `8192` | ≥10 g |
-| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g |
+| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) |
+| ----------------------------------- | ------------------------------- | ---------------------- |
+| Dashboard / light chat | `1024` (image default) | ≥2 g |
+| One coding agent | `8192` | ≥10 g |
+| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g |
```bash
docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \
@@ -1003,6 +1003,7 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \
```
Full table: [Docker Guide — runtime RAM](docs/guides/DOCKER_GUIDE.md#runtime-ram-for-coding-agents).
+
> **Pre-release Docker channel:** `diegosouzapw/omniroute:next` and
> `diegosouzapw/omniroute:next-web` follow the current default `release/v*`
> branch. These mutable tags are intended only for testing unreleased fixes and
diff --git a/bin/cli/commands/combo.mjs b/bin/cli/commands/combo.mjs
index 554c1c1d838..8d58cf73bd9 100644
--- a/bin/cli/commands/combo.mjs
+++ b/bin/cli/commands/combo.mjs
@@ -307,6 +307,12 @@ export async function runComboCreateCommand(name, strategy = "priority", opts =
}
const models = Array.isArray(opts.models) ? opts.models : [];
+ if (!models.length) {
+ console.error(
+ "combo create requires at least one target. Pass --models and/or repeat --model ."
+ );
+ return 1;
+ }
try {
return await withRuntime(async ({ kind, api, db }) => {
diff --git a/changelog.d/features/10987-logfare-free-provider.md b/changelog.d/features/10987-logfare-free-provider.md
new file mode 100644
index 00000000000..507a528411f
--- /dev/null
+++ b/changelog.d/features/10987-logfare-free-provider.md
@@ -0,0 +1 @@
+- **feat(providers):** add Logfare as a free OpenAI-compatible provider — dashboard card with a Free badge and request-logging disclosure (every prompt/completion is logged for research; opt out at logfare.ai/consent), live model discovery from `https://logfare.ai/v1/models` (20 models, 11 chat-capable: kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3…), full chat/streaming through the existing OpenAI-compatible path, the real Logfare logo on the card, and a listing in the free-tiers guide. ([#10987](https://github.com/diegosouzapw/OmniRoute/pull/10987))
diff --git a/changelog.d/features/11190-usage-command-json.md b/changelog.d/features/11190-usage-command-json.md
new file mode 100644
index 00000000000..d7655f04c51
--- /dev/null
+++ b/changelog.d/features/11190-usage-command-json.md
@@ -0,0 +1 @@
+- **feat(api):** `/api/usage/om-usage` gains a structured form — `?format=json` returns the key's own usage as `ApiKeyUsageLimitStatus` + `UsageSnapshot` instead of `text/plain`. This is the surface a UI (the OmniCopilot panel) consumes to show a key holder their daily/weekly spend and quota reset. The route is self-service (the caller's own key, gated by `allowUsageCommand`), not the management surface; refusals come back as a discriminated `{ "allowed": false, "error": … }` so a UI can tell "not allowed" apart from "allowed but nothing cached yet". The endpoint was previously undocumented in `API_REFERENCE.md`; it now has a section ([#11190](https://github.com/diegosouzapw/OmniRoute/pull/11190))
diff --git a/changelog.d/features/11192-usage-command-providers-array.md b/changelog.d/features/11192-usage-command-providers-array.md
new file mode 100644
index 00000000000..b7ef4211091
--- /dev/null
+++ b/changelog.d/features/11192-usage-command-providers-array.md
@@ -0,0 +1 @@
+- **feat(api):** `/api/usage/om-usage?format=json` now returns `providers[]` — every connection's quota snapshot, not just the single selected one — so a panel can render Codex / Claude / OpenCode side by side. The collector already gathered all of them; the single-pick `provider` field (kept) is a terminal presentation choice. Closes the per-connection gap from OmniCopilot #8 ([#11192](https://github.com/diegosouzapw/OmniRoute/pull/11192))
diff --git a/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md b/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md
new file mode 100644
index 00000000000..47f8de84fd2
--- /dev/null
+++ b/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md
@@ -0,0 +1 @@
+- **fix(providers):** the five g4f.space sub-providers (Groq, Gemini, Pollinations, Ollama, NVIDIA) no longer advertise a free tier — a keyless `POST /v1/chat/completions` now returns `402 insufficient_credits` behind a proof-of-work "cake" wall (re-verified live 2026-08-22), so `hasFree` is `false` and the notes point at `g4f.dev/members.html`. The gateway still works with a member key, so its registry wiring and `authType: "optional"` are unchanged ([#10071](https://github.com/diegosouzapw/OmniRoute/issues/10071)) — thanks @chirag127
diff --git a/changelog.d/fixes/10550-responses-reasoning-transport.md b/changelog.d/fixes/10550-responses-reasoning-transport.md
index d34c433debd..e2b40cdb8c9 100644
--- a/changelog.d/fixes/10550-responses-reasoning-transport.md
+++ b/changelog.d/fixes/10550-responses-reasoning-transport.md
@@ -1 +1 @@
-- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Combos now drop incompatible continuation reasoning by default and can explicitly skip incompatible targets, while known providers no longer show redundant encrypted-reasoning controls. (#10550)
+- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Direct requests drop incompatible continuation reasoning by default; combos can explicitly skip incompatible targets without mutating the request. Known providers no longer show redundant encrypted-reasoning controls. (#10550, #10959)
diff --git a/changelog.d/fixes/10949-mixed-reasoning-plaintext.md b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md
new file mode 100644
index 00000000000..05a055ec538
--- /dev/null
+++ b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md
@@ -0,0 +1 @@
+- Preserve explicit plaintext reasoning when a Responses reasoning item also carries opaque provider state (rare OpenCode Go `deepseek-v4-flash` responses). Mixed plaintext + opaque input is projected onto the target transport: plaintext targets keep portable text, opaque targets keep provider state. Opaque-only reasoning is dropped when the selected target cannot replay it, allowing cross-model conversations to continue. (#10949, #10959)
diff --git a/changelog.d/fixes/11015-shutdown-track-sse.md b/changelog.d/fixes/11015-shutdown-track-sse.md
new file mode 100644
index 00000000000..1ed99b3669d
--- /dev/null
+++ b/changelog.d/fixes/11015-shutdown-track-sse.md
@@ -0,0 +1 @@
+- **fix(resilience):** count heavyweight `/v1` admission leases in the SIGTERM drain and send `Retry-After` on shutdown 503s so Recreate no longer looks like an empty 502 ([#11015](https://github.com/diegosouzapw/OmniRoute/issues/11015)) — thanks @RaviTharuma
diff --git a/changelog.d/fixes/11089-chat-routing-synced-inventory.md b/changelog.d/fixes/11089-chat-routing-synced-inventory.md
new file mode 100644
index 00000000000..922b96a659b
--- /dev/null
+++ b/changelog.d/fixes/11089-chat-routing-synced-inventory.md
@@ -0,0 +1 @@
+- **fix(resilience):** filter chat connection selection by each connection's *synced* model inventory on multi-host self-hosted providers (`ollama-local`, `lm-studio`, `vllm`, …), so a request for a model only one host advertises is pinned to that host instead of failing over onto a host that never had it ([#11089](https://github.com/diegosouzapw/OmniRoute/issues/11089))
diff --git a/changelog.d/fixes/11149-opencode-go-flat-rate.md b/changelog.d/fixes/11149-opencode-go-flat-rate.md
new file mode 100644
index 00000000000..7aa63ad4559
--- /dev/null
+++ b/changelog.d/fixes/11149-opencode-go-flat-rate.md
@@ -0,0 +1 @@
+- **fix(analytics):** `opencode-go` is now classified as a flat-rate subscription, so cost analytics shows $0 for it instead of billing every call at the underlying model’s metered rate — it resells GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x under one flat monthly fee, which made the overstatement large rather than marginal ([#11149](https://github.com/diegosouzapw/OmniRoute/pull/11149)) — thanks @electrumguy
diff --git a/changelog.d/fixes/11162-combo-create-requires-model.md b/changelog.d/fixes/11162-combo-create-requires-model.md
new file mode 100644
index 00000000000..228e6a9b20f
--- /dev/null
+++ b/changelog.d/fixes/11162-combo-create-requires-model.md
@@ -0,0 +1 @@
+- **Combo create:** creating a routing combo without any model is now refused (`400`) — the CLI requires `--models`/`--model` on `combo create`, matching the dashboard which already rejected empty combos.
diff --git a/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md b/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md
new file mode 100644
index 00000000000..c533feae543
--- /dev/null
+++ b/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md
@@ -0,0 +1 @@
+- **fix(routing):** a custom `openai-compatible-*` / `anthropic-compatible-*` connection pointing at a keyless self-hosted backend (llama.cpp, Ollama, vLLM started without an API key) now stays in the `auto/*` candidate pool instead of being silently dropped by the credential gate — for those IDs "no credential" is the normal configuration, not an unconfigured connection ([#11180](https://github.com/diegosouzapw/OmniRoute/pull/11180)) — thanks @marcs7
diff --git a/changelog.d/fixes/11181-lkgp-enabled-context.md b/changelog.d/fixes/11181-lkgp-enabled-context.md
new file mode 100644
index 00000000000..d1c0cde5a36
--- /dev/null
+++ b/changelog.d/fixes/11181-lkgp-enabled-context.md
@@ -0,0 +1 @@
+- **fix(routing):** the Routing tab's "last known good provider" toggle now actually takes effect — `lkgpEnabled` was persisted and the `lkgp` strategy guarded on it, but the setting was never forwarded into the `RoutingContext` built in `resolveAutoStrategyOrder()`, so `context.lkgpEnabled` was always `undefined` and the off-switch was unreachable ([#11181](https://github.com/diegosouzapw/OmniRoute/issues/11181))
diff --git a/changelog.d/fixes/9763-ratelimit-mintime-floor.md b/changelog.d/fixes/9763-ratelimit-mintime-floor.md
new file mode 100644
index 00000000000..2b145f1e1a3
--- /dev/null
+++ b/changelog.d/fixes/9763-ratelimit-mintime-floor.md
@@ -0,0 +1 @@
+- **fix(ratelimit):** respect operator `minTimeBetweenRequestsMs` floor when relaxing the limiter on headroom — the adaptive rate-limit learning no longer silently erases a configured minimum gap between requests when the upstream reports plenty of remaining capacity ([#9763](https://github.com/diegosouzapw/OmniRoute/issues/9763)).
diff --git a/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md
new file mode 100644
index 00000000000..82a88c5905a
--- /dev/null
+++ b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md
@@ -0,0 +1 @@
+- **fix(executors):** OpencodeExecutor rotates (or retries once on a single-account direct path) on upstream 400 empty-body rejections — malformed completion envelopes with no error field were propagated as success and killed client sessions. Bounded +1 attempt per request; body reads are conditioned on status 400 so successful/streaming responses are never buffered. 400s carrying an error field keep propagating immediately.
diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json
index 3dae8591dd7..79875e31489 100644
--- a/config/quality/eslint-suppressions.json
+++ b/config/quality/eslint-suppressions.json
@@ -853,11 +853,6 @@
"count": 1
}
},
- "src/app/api/usage/call-logs/route.ts": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/app/api/usage/quota/route.ts": {
"no-restricted-imports": {
"count": 1
@@ -953,11 +948,6 @@
"count": 1
}
},
- "src/app/api/v1/rerank/route.ts": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/app/api/v1/vscode/[token]/models/route.ts": {
"no-restricted-syntax": {
"count": 1
@@ -3259,4 +3249,4 @@
"count": 5
}
}
-}
+}
\ No newline at end of file
diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json
index b7f40cebfa6..0a3239ea7fb 100644
--- a/config/quality/file-size-baseline.json
+++ b/config/quality/file-size-baseline.json
@@ -1,5 +1,6 @@
{
"_rebaseline_2026_08_20_10531_freebuff_provider": "PR #10531 (adrianaryaputra, feat/freebuff-provider-support, closes #6793) own growth: src/shared/constants/providers/apikey/gateways.ts 1283->1298 (+15, the freebuff APIKEY_PROVIDERS_GATEWAYS catalog entry, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines) and src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx 1062->1067 (+5, freebuff credential placeholder/hint at the existing per-provider switch chokepoint). Covered by tests/unit/freebuff-provider.test.ts (9/9 passing).",
+ "_rebaseline_2026_08_21_10987_logfare_provider": "PR #10987 (jonlwheat2-gif, feat/10644-logfare-provider, closes #10644) own growth: src/shared/constants/providers/apikey/gateways.ts 1298->1321 (+23, the logfare APIKEY_PROVIDERS_GATEWAYS catalog entry with Free badge/freeNote/apiHint documenting the request-logging policy, additive data at the existing registry chokepoint, same god-file no-split rationale as the prior gateways.ts rebaselines: #10531 freebuff, merge-storm 2026-08-11). Covered by tests/unit/logfare-registry.test.ts (1/1 passing).",
"_rebaseline_2026_08_20_10574_reasoning_transport_fallback": "PR #10574 (jackjinke, fix/responses-reasoning-transport, fixes #10550) own growth: src/sse/handlers/chatHelpers.ts 1017->1019 (+2 = the new reasoningTransportFallback option threaded through executeChatWithBreaker's options destructure and its downstream handleSingleModel call, at the existing per-attempt options-passthrough chokepoint; not extractable without splitting the option-forwarding call itself). Covered by the PR's own reasoning-policy test suite (tests/unit/chatcore-translation-paths.test.ts, tests/unit/combo-attempt-body-isolation-7847.test.ts, tests/unit/reasoning-cache.test.ts, tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts among others), 446/446 focused tests passing.",
"_rebaseline_2026_08_18_10517_zed_hosted_oauth_callback_port": "PR #10517 (phatchau036, fix/zed-hosted-oauth-callback-port) own growth: src/shared/components/OAuthModal.tsx 1131->1148 (wc -l; check-file-size.mjs counts via split(\"\\n\").length so the gate sees 1134->1149, +15/+18, crosses the frozen 1134 cap). Wires the zed-hosted native-app callback auto-complete: forceManual gating on isTrueLocalhost for zed-hosted, the loopback-redirect-URI comment block, and the exchangeToken full-URL-as-code branch, all at the existing provider-switch chokepoints this modal already carries growth for (seventh bump: 969->989->993->998->1030->1056->1100->1149; structural shrink tracked in #3501). The actual port-derivation logic lives in src/lib/oauth/providers/zed-hosted.ts (not frozen here) and was hardened during pre-merge review to use the server's own getRuntimePorts() instead of a browser-guessed scheme/port, covered by the new tests/unit/zed-hosted-loopback-port-derivation.test.ts (8/8 passing).",
"_rebaseline_2026_08_13_10243_codex_fingerprint_merge": "PR #10243 (xz-dev, Codex OAuth fingerprint convergence) merge into release/v3.8.50: src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts crossed the 1000-line new-file cap for the first time (974 on base, 997 on the PR's own branch, 1013 after merging + prettier reflow) purely from combining two independent, already-legitimate feature additions that landed on the same shared UI-helper file — this PR's own Codex fingerprint-mode select/toggle wiring (CODEX_FINGERPRINT_MODE_VALUES, getCodexFingerprintModeLabel, CodexFingerprintModeValue) plus #8949's unrelated Codex account-service-tier helpers merged concurrently on release/v3.8.50. Neither addition alone crosses the cap; git's line-level auto-merge does not detect a threshold crossing. Not modularized as part of this conflict-resolution merge commit (out of scope — this is a merge, not a feature change). Covered by the PR's own tests/unit/codex-fingerprint-convergence.test.ts, tests/unit/executor-codex.test.ts, tests/unit/provider-specific-data-schema.test.ts (all passing post-merge).",
@@ -388,6 +389,10 @@
"open-sse/services/claudeCodeCompatible.ts": 1563,
"open-sse/services/combo.ts": 4742,
"open-sse/services/compression/strategySelector.ts": 1379,
+ "open-sse/services/compression/engines/ccr/index.ts": 1024,
+ "_rebaseline_2026_08_22_11084_ccr_caller_gate": "PR #11084 (HouMinXi) own growth: open-sse/services/compression/engines/ccr/index.ts 1000->1024 (first listing — the engine was unlisted and drifted just over the 1000 cap; +24 are the callerSupportsCcrRetrieve gate that skips replacement entirely for callers without the retrieve tool, closing the stranded-prompt incident measured in production). Covered by tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts. Owner pre-authorized baseline bumps 2026-08-22.",
+ "open-sse/services/contextManager.ts": 1001,
+ "_rebaseline_2026_08_22_11113_purify_system_first": "PR #11113 (ggdayup) own growth: open-sse/services/contextManager.ts 1000->1001 (+1, purifyHistory merges the compression notice into the leading system message instead of splicing a second one mid-array — live-confirmed TokenRouter 400s; the +1 is the merge-into-leading branch, not extractable). Covered by tests/unit/context-manager-purify-system-first.test.ts. Owner pre-authorized baseline bumps 2026-08-22.",
"open-sse/services/rateLimitManager.ts": 1517,
"open-sse/translator/response/openai-responses.ts": 1652,
"open-sse/utils/cursorAgentProtobuf.ts": 1956,
@@ -427,7 +432,8 @@
"src/shared/components/analytics/charts.tsx": 1346,
"src/shared/services/cliRuntime.ts": 1459,
"src/sse/handlers/chat.ts": 2493,
- "src/sse/services/auth.ts": 3260,
+ "src/sse/services/auth.ts": 3337,
+ "_rebaseline_2026_08_23_11186_synced_inventory_routing": "PR #11186 (pacocartones) own growth: src/sse/services/auth.ts 3260->3337 (+77, loadAdvertisedModelsForSelfHostedConnections + the modelNotAdvertised candidate-filter predicate — pins chat routing to the connection whose synced inventory actually advertises the model, fixing spurious model-not-found on multi-host self-hosted setups; at the existing credential-selection chokepoint, not extractable without splitting the selection flow). Covered by tests/unit/chat-routing-synced-inventory-11089.test.ts. Owner pre-authorized baseline bumps 2026-08-22.",
"tests/unit/account-fallback-service.test.ts": 2044,
"tests/unit/provider-validation-specialty.test.ts": 3880,
"open-sse/executors/hyperagent.ts": 1334,
@@ -436,7 +442,8 @@
"open-sse/executors/kiro.ts": 1390,
"open-sse/translator/request/openai-to-kiro.ts": 1374,
"open-sse/utils/sseHeartbeat.ts": 194,
- "open-sse/utils/proxyFetch.ts": 1239,
+ "open-sse/utils/proxyFetch.ts": 1244,
+ "_rebaseline_2026_08_23_11177_dns_retry_classification": "PR #11177 (rqzbeh) own growth: proxyFetch.ts 1239->1244 (+5, EAI_AGAIN/ENOTFOUND/ETIMEDOUT join the retryable dispatcher classification alongside ECONNREFUSED — bounded socket retries for transient DNS failures, part of the #10443 Hermes→Antigravity stream-drop fixes). Covered by tests/unit/proxy-fetch-dns-retry-10443.test.ts. Owner pre-authorized baseline bumps 2026-08-22.",
"_rebaseline_2026_08_11_v3850_merge_storm_provider_registry: DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legítima acima do cap; gateways.ts = god-file de catálogo de providers que cresceu com os PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o próprio PR #9421 foi o que quebrou o arquivo; sem split até o release, congelado no tamanho atual). Owner autorizou rebaseline com anotação (2026-08-11).": {
"src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1062,
"src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051,
@@ -447,7 +454,7 @@
"_rebaseline_2026_08_22_11156_enter_check_disabled": "PR #11156 (rqzbeh) own growth: AddApiKeyModal.tsx 1080->1082 (+2, Enter keydown handler now mirrors the isCheckDisabled condition — owner-requested post-merge polish from #11056; the rest of the diff is Prettier reflow). Covered by tests/unit/ui/add-api-key-modal-enter-key.test.tsx (jsdom render test, Enter dispatch assertions).",
"src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051,
"src/shared/components/ModelSelectModal.tsx": 1138,
- "src/shared/constants/providers/apikey/gateways.ts": 1298,
+ "src/shared/constants/providers/apikey/gateways.ts": 1321,
"open-sse/vendor/codex-chatgpt-web/bridge.ts": 1387,
"_rebaseline_2026_08_11_v3850_merge_storm_provider_registry": "DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legitima acima do cap; gateways.ts = god-file de catalogo de providers que cresceu com PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o proprio PR #9421 quebrou o arquivo); bridge.ts (PR #8949) = ponte Chromium vendored; proxyFetch.ts 1207->1220 = drift herdado de merges. Owner autorizou rebaseline com anotacao (2026-08-11).",
"src/lib/modelCapabilities.ts": 1072,
@@ -455,7 +462,8 @@
"src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 1014,
"open-sse/config/imageRegistry.ts": 1034,
"src/sse/handlers/chatHelpers.ts": 1019,
- "src/shared/middleware/chatBodyAdmission.ts": 1005,
+ "src/shared/middleware/chatBodyAdmission.ts": 1009,
+ "_rebaseline_2026_08_22_11020_sigterm_drain": "PR #11020 (RaviTharuma) own growth: chatBodyAdmission.ts 1005->1009 (+4, heavyweight admission leases now increment the SIGTERM drain counter and releaseChatAdmissionWhenDone holds it for the SSE lifetime — closes #11015; +4 are the lease/drain wiring lines at the existing admission chokepoint). Covered by tests/unit/chat-body-admission.test.ts heavyweight-lease cases. Owner pre-authorized baseline bumps 2026-08-22.",
"_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file).",
"open-sse/executors/commandCode.ts": 1059,
"_rebaseline_2026_08_21_10859_vision_bridge_catalog": "#10859 own growth (Vision Bridge fixes #10808/#10809): src/lib/modelCapabilities.ts 1006->1016 (+10, cmd/gpt-5.3-codex* text-only capability resolution) and open-sse/executors/commandCode.ts 988->1023 (+35, Command Code wire-model normalization for bare ids + reasoning field fallback for opencode-routed gateways). Cohesive bug fixes at the existing capability-resolution / executor chokepoints; not extractable mid-fix. Covered by tests/unit/model-capabilities-command-code-codex-textonly-10703.test.ts, tests/unit/command-code-vision.test.ts, tests/unit/opencode-mimo-reasoning-details-nonstream.test.ts. Pushed directly to release (own-session miss: the original rebaseline was made in a throwaway validation worktree and never landed on the PR branch or the release before merge).",
diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json
index 95e35a5adec..604a5a8e19d 100644
--- a/config/quality/quality-baseline.json
+++ b/config/quality/quality-baseline.json
@@ -197,10 +197,11 @@
"_rebaseline_2026_08_09_v3850_release_close": "7666 -> 8045 (+379 gzip bytes, +4.9%). Release v3.8.50 close reconciliation measured twice with the real size-limit + @size-limit/file path on tip e0ce95c592. Per-entry measurements remain below their absolute budgets: omniroute.mjs 4380/15000, mcp-server.mjs 1195/5000, nodeRuntimeSupport.mjs 887/8000, reset-password.mjs 1583/6000. The growth accumulated through legitimate CLI/runtime work in this cycle, including global-install ESM alias resolution, Termux cache preparation, and MCP stdio startup hardening; no entrypoint is near its absolute ceiling. The direction:down ratchet stays blocking from this exact measured tip."
},
"openapiBreaking": {
- "value": 0,
+ "value": 4,
"direction": "down",
"dedicatedGate": true,
- "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec."
+ "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec.",
+ "_rebaseline_2026_08_22_combo_create_min1": "0 -> 4, split 3 own + 1 inherited. Docs-only alignment of components.schemas.ComboCreate with the request contract already enforced by the API since 638fc5fbd (combo create refuses an empty model list) and d5034ea52: `model`/`nodes` were phantom properties the server never accepted, and `models` (array, minItems 1) is the real required field. OWN findings (3, caused by this commit): removed `model`, removed `nodes`, added required `models` on POST /api/combos — spec-vs-server drift, not client-facing breakage, no working client could have relied on the removed shapes. INHERITED finding (1, NOT caused by this PR's code changes — pre-existing drift already present at parent d5034ea52): PATCH /api/combos/{id} request-body-added-required; that route's patch operation declares its own inline requestBody (required: true, bare object schema, docs/openapi.yaml ~2107-2118) and does not reference ComboCreate, so this finding exists independently of the ComboCreate alignment (same own-growth vs inherited-drift convention as _rebaseline_2026_07_20_aliasresolver_hook_split_7808). No code change in this PR; follow-up tracking = this change's PR description."
},
"mutationScore.src/sse/services/auth.ts": {
"value": 52.57,
diff --git a/docs/changelog/fragments/10962.md b/docs/changelog/fragments/10962.md
new file mode 100644
index 00000000000..5170a415e3f
--- /dev/null
+++ b/docs/changelog/fragments/10962.md
@@ -0,0 +1 @@
+fix(catalog): expose only provider-routable GLM reasoning-effort tiers and remove unroutable ZCode aliases
diff --git a/docs/diagrams/cli-terminal.svg b/docs/diagrams/cli-terminal.svg
index e2ad57c8b13..6ec68793d0c 100644
--- a/docs/diagrams/cli-terminal.svg
+++ b/docs/diagrams/cli-terminal.svg
@@ -1,4 +1,4 @@
-
import("./components/AddCompatibleProviderModal"),
+ { ssr: false }
+);
import { CategoryDot } from "./components/CategoryDot";
-import { ImportProvidersFromFileModal } from "./components/ImportProvidersFromFileModal";
+const ImportProvidersFromFileModal = dynamic(
+ () =>
+ import("./components/ImportProvidersFromFileModal").then(
+ (m) => m.ImportProvidersFromFileModal
+ ),
+ { ssr: false }
+);
import NoAuthProvidersSection from "./components/NoAuthProvidersSection";
import HighlightableProviderCard from "./components/HighlightableProviderCard";
import ProviderCountBadge from "./components/ProviderCountBadge";
diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx
index 3af32da285d..dc102936d09 100644
--- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx
+++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx
@@ -98,7 +98,7 @@ export function formatQuotaLabel(name: string) {
return `Weekly ${toTitleCaseWords(weeklyModelMatch[1])}`;
}
- return trimmed;
+ return toTitleCaseWords(trimmed.replace(/_/g, " "));
}
/**
diff --git a/src/app/(dashboard)/home/page.tsx b/src/app/(dashboard)/home/page.tsx
index beccde22610..bc10df88f42 100644
--- a/src/app/(dashboard)/home/page.tsx
+++ b/src/app/(dashboard)/home/page.tsx
@@ -4,6 +4,7 @@ import { getSettings } from "@/lib/localDb";
import HomePageClient from "../dashboard/HomePageClient";
import BootstrapBanner from "../dashboard/BootstrapBanner";
import KimiSponsorBanner from "../dashboard/KimiSponsorBanner";
+import CheaperInferenceSponsorBanner from "../dashboard/CheaperInferenceSponsorBanner";
import VscodeCopilotBanner from "../dashboard/VscodeCopilotBanner";
import NewsBanner from "../dashboard/NewsBanner";
@@ -20,6 +21,7 @@ export default async function HomePage() {
<>
{isBootstrapped && }
+
diff --git a/src/app/api/providers/[id]/models/discovery/providerSets.ts b/src/app/api/providers/[id]/models/discovery/providerSets.ts
index 82386814790..2b35e54b01e 100644
--- a/src/app/api/providers/[id]/models/discovery/providerSets.ts
+++ b/src/app/api/providers/[id]/models/discovery/providerSets.ts
@@ -95,6 +95,11 @@ export const NAMED_OPENAI_STYLE_PROVIDERS = new Set([
"internlm",
"ant-ling",
"nanogpt",
+ // Logfare (https://logfare.ai) — free OpenAI-compatible gateway live-verified
+ // 2026-08-21: GET https://logfare.ai/v1/models returns a real 20-model catalog
+ // (11 chat-capable). Live fetch keeps it fresh; the registry seed stays as the
+ // offline fallback.
+ "logfare",
]);
export function isNamedOpenAIStyleProvider(provider: string): boolean {
diff --git a/src/app/api/providers/[id]/models/route.ts b/src/app/api/providers/[id]/models/route.ts
index 01f1bd72857..cecbb6776eb 100755
--- a/src/app/api/providers/[id]/models/route.ts
+++ b/src/app/api/providers/[id]/models/route.ts
@@ -1909,12 +1909,9 @@ export async function GET(
}
if (isAnthropicCompatibleProvider(provider)) {
- const cachedResponse = maybeReturnCachedDiscovery();
- if (cachedResponse) return cachedResponse;
-
- const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled();
- if (autoFetchDisabledResponse) return autoFetchDisabledResponse;
-
+ // CC providers never support models listing — this check must precede
+ // the cached-discovery / auto-fetch fallbacks, which would otherwise
+ // return a misleading 200 "no models" for a CC node (#10828 ordering).
if (isClaudeCodeCompatibleProvider(provider)) {
return NextResponse.json(
{ error: `Provider ${provider} does not support models listing` },
@@ -1922,6 +1919,12 @@ export async function GET(
);
}
+ const cachedResponse = maybeReturnCachedDiscovery();
+ if (cachedResponse) return cachedResponse;
+
+ const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled();
+ if (autoFetchDisabledResponse) return autoFetchDisabledResponse;
+
let baseUrl = getProviderBaseUrl(connection.providerSpecificData);
if (!baseUrl) {
const fallback = buildDiscoveryFallbackResponse({
diff --git a/src/app/login/page.tsx b/src/app/login/page.tsx
index f2c70306f38..664eb3c1721 100644
--- a/src/app/login/page.tsx
+++ b/src/app/login/page.tsx
@@ -38,8 +38,7 @@ export default function LoginPage() {
if (data.nodeVersion) setNodeVersion(data.nodeVersion);
if (data.nodeCompatible === false) setNodeCompatible(false);
if (data.authenticated === true || data.requireLogin === false) {
- router.push("/dashboard");
- router.refresh();
+ window.location.href = "/dashboard";
return;
}
setHasPassword(!!data.hasPassword);
@@ -77,13 +76,12 @@ export default function LoginPage() {
if (res.ok) {
sessionStorage.setItem("omniroute_login_time", String(Date.now()));
- router.push("/dashboard");
- router.refresh();
+ window.location.href = "/dashboard";
} else {
const data = await res.json();
// (#521) If no password is set, redirect to onboarding instead of showing an error
if (data.needsSetup) {
- router.push("/dashboard/onboarding");
+ window.location.href = "/dashboard/onboarding";
return;
}
setError(data.error || t("invalidPassword"));
diff --git a/src/domain/connectionModelRules.ts b/src/domain/connectionModelRules.ts
index 316ade72d80..7831bbc84b6 100644
--- a/src/domain/connectionModelRules.ts
+++ b/src/domain/connectionModelRules.ts
@@ -80,3 +80,26 @@ export function hasEligibleConnectionForModel(
(connection) => !isModelExcludedByConnection(modelId, connection?.providerSpecificData)
);
}
+
+/**
+ * #11089: does this connection's *synced* inventory advertise the model?
+ *
+ * Unlike `excludedModels` (a manually maintained denylist) this reads the
+ * per-connection catalog written by model discovery, so a multi-host local
+ * provider never routes a model to a host that never had it. Ids are matched
+ * with the same candidate semantics as the denylist (provider prefix and the
+ * `[1m]` extended-context suffix are tolerated), but never as wildcard
+ * patterns — a synced id is a literal.
+ *
+ * Fails OPEN on an empty inventory: a host that has not been synced yet is
+ * "unknown", not "does not have it".
+ */
+export function isModelAdvertisedByConnection(
+ modelId: unknown,
+ advertisedModelIds: ReadonlySet | null | undefined
+): boolean {
+ if (!advertisedModelIds || advertisedModelIds.size === 0) return true;
+ if (typeof modelId !== "string" || modelId.trim().length === 0) return true;
+
+ return getModelMatchCandidates(modelId).some((candidate) => advertisedModelIds.has(candidate));
+}
diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json
index ff778753c61..930f5b59587 100644
--- a/src/i18n/messages/en.json
+++ b/src/i18n/messages/en.json
@@ -13869,5 +13869,12 @@
"toolsMismatch": "Provider does not support tool calling",
"structuredOutputMismatch": "Provider does not support structured output",
"contextWindowMismatch": "Request exceeds provider context window"
+ },
+ "cheaperInferenceSponsorBanner": {
+ "title": "Cheaper Inference is an OmniRoute Open Source Friend",
+ "description": "A cost-ranked gateway reselling dozens of frontier models behind one OpenAI-compatible endpoint — routing each request to the cheapest eligible provider, never above list price.",
+ "cta": "Get an API Key",
+ "partnerLinkNote": "Partner link",
+ "dismissAriaLabel": "Dismiss"
}
}
diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json
index f3e3c15d130..8df20f0f9df 100644
--- a/src/i18n/messages/pt-BR.json
+++ b/src/i18n/messages/pt-BR.json
@@ -4922,7 +4922,9 @@
"multiProvider": "Multi-Provedor",
"usageTracking": "Rastreamento de Uso",
"securityDesc": "Defina uma senha para proteger seu painel, ou pule por enquanto.",
+ "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você poderá adicioná-los depois pelo painel, após definir uma senha.",
"providerDesc": "Conecte seu primeiro provedor de IA. Você pode adicionar mais depois.",
+ "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois pelo painel.",
"apiKeyRequired": "Chave de API (obrigatório)",
"customUrlOptional": "URL personalizada (opcional)",
"testDesc": "Vamos verificar se a conexão com seu provedor funciona.",
@@ -4979,9 +4981,7 @@
"skipped": "já configurado",
"failed": "falhou"
}
- },
- "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você pode adicioná-los depois no painel após definir uma senha.",
- "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois no painel."
+ }
},
"providers": {
"title": "Provedores",
@@ -6269,6 +6269,20 @@
"webSessionGuideStep3": "Copie a credencial necessária do próprio domínio do provedor. Para cookies, copie apenas o valor do cabeçalho Cookie e omita Cookie:.",
"webSessionGuideStep3Manual": "Caminho manual: abra as ferramentas do desenvolvedor do navegador (F12 → Network), atualize a página, abra uma requisição autenticada e copie o valor do cabeçalho Cookie em Request Headers — omita o prefixo Cookie:.",
"webSessionGuideStep4": "Cole aqui e verifique a conexão. Se parar de funcionar, faça login novamente e substitua-o por um novo valor.",
+ "harImportButtonLabel": "Importar arquivo .har",
+ "harImportButtonBusy": "Importando…",
+ "harImportButtonHint": "Exporte pela aba Rede das Ferramentas do Desenvolvedor após enviar pelo menos uma mensagem no chat.",
+ "harImportStatusValid": "Importado — válido por cerca de {minutes} min.",
+ "harImportStatusExpiringSoon": "Importado — válido por apenas mais cerca de {minutes} min.",
+ "harImportStatusExpired": "Importado, mas este token expirou há {minutes} min — exporte um HAR novo.",
+ "harImportStatusUnknownExpiry": "Importado. Não foi possível ler a expiração.",
+ "harImportErrorNotJson": "Esse arquivo não é um JSON válido — ele é realmente uma exportação .har?",
+ "harImportErrorNoEntries": "Este HAR não contém entradas de rede.",
+ "harImportErrorNoChathubUrl": "Nenhuma conexão de chat do Copilot foi encontrada neste HAR. Envie pelo menos uma mensagem em m365.cloud.microsoft antes de exportar.",
+ "harImportErrorUnparsableUrl": "A conexão de chat foi encontrada, mas não foi possível ler a URL.",
+ "harImportErrorMissingFields": "A conexão de chat foi encontrada, mas o token estava ausente.",
+ "harImportErrorReadFailed": "Não foi possível ler esse arquivo.",
+ "harImportErrorUnknown": "Não foi possível extrair uma credencial desse arquivo HAR.",
"webSessionSecurityHint": "Trate isso como uma senha: ela poderá acessar sua conta da web conectada até que ela expire ou seja revogada.",
"webNoAuthGuideTitle": "Nenhuma credencial necessária",
"webNoAuthGuideBody": "{provider} não precisa de chave de API ou cookie. Salve a conexão para usar seu endpoint web gratuito.",
diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json
index 11a3606a54c..1921f3fba7f 100644
--- a/src/i18n/messages/pt.json
+++ b/src/i18n/messages/pt.json
@@ -13845,5 +13845,12 @@
"toolsMismatch": "Provider does not support tool calling",
"structuredOutputMismatch": "Provider does not support structured output",
"contextWindowMismatch": "Request exceeds provider context window"
+ },
+ "cheaperInferenceSponsorBanner": {
+ "title": "A Cheaper Inference é uma Amiga do Código Aberto do OmniRoute",
+ "description": "Um gateway com custo ordenado que revende dezenas de modelos de fronteira num único endpoint compatível com OpenAI — roteando cada requisição ao provedor elegível mais barato, nunca acima do preço de tabela.",
+ "cta": "Obter uma Chave de API",
+ "partnerLinkNote": "Link de parceiro",
+ "dismissAriaLabel": "Dispensar"
}
}
diff --git a/src/lib/memory/injection.ts b/src/lib/memory/injection.ts
index 9fdd2bbbf10..d4d8ead7f7c 100644
--- a/src/lib/memory/injection.ts
+++ b/src/lib/memory/injection.ts
@@ -65,7 +65,11 @@ export function providerSupportsSystemMessage(provider: string | null | undefine
*
* Populated with the Xiaomi MiMo endpoint (provider id `xiaomi-mimo`, registry
* alias `mimo`, serving mimo-v2.5) confirmed live to 400 on a non-first system
- * message. Add other providers here only when they are documented as strict.
+ * message, and the TokenRouter gateway (provider id `tokenrouter`), confirmed
+ * live on 2026-08-22 to reject mid-array system messages — including the
+ * compression notice spliced by purifyHistory() before that splice was fixed to
+ * merge into the leading system message. Add other providers here only when
+ * they are documented as strict.
*
* Self-hosted deployments can extend this list without a source change via
* OMNIROUTE_STRICT_SYSTEM_PROVIDERS (comma-separated provider ids,
@@ -73,7 +77,7 @@ export function providerSupportsSystemMessage(provider: string | null | undefine
* self-hosted Qwen3.5+/3.6 model, whose chat template enforces the same
* single-leading-system-message constraint as xiaomi-mimo.
*/
-const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo"]);
+const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo", "tokenrouter"]);
/**
* Parses OMNIROUTE_STRICT_SYSTEM_PROVIDERS into a normalized id list.
diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts
index 3828047b90b..b7aa2aa2164 100644
--- a/src/lib/modelMetadataRegistry.ts
+++ b/src/lib/modelMetadataRegistry.ts
@@ -130,6 +130,11 @@ function uniqueStrings(values: Array) {
];
}
+export function isGlmFamilyModel(modelId: string, displayName = ""): boolean {
+ const glmFamilyPattern = /(?:^|[/@:_. -])glm(?=$|[-._ /@:](?:z)?\d|\d)/i;
+ return glmFamilyPattern.test(modelId) || glmFamilyPattern.test(displayName);
+}
+
function toQualifiedId(
providerAlias: string | null,
provider: string | null,
@@ -477,11 +482,16 @@ export function enrichCatalogModelEntry(
? declaredEffortTiers
: sourceDeclaresThinking
? undefined
- : extendCodexGpt56EffortValues(
- metadata.provider,
- metadata.model,
- CANONICAL_EFFORT_VALUES
- );
+ : // #10963: GLM-family models never inherit generic OpenAI tiers — an
+ // explicit empty list is authoritative unless a provider-declared
+ // contract exists (handled by declaredEffortTiers above).
+ isGlmFamilyModel(metadata.model, metadata.displayName)
+ ? []
+ : extendCodexGpt56EffortValues(
+ metadata.provider,
+ metadata.model,
+ CANONICAL_EFFORT_VALUES
+ );
const capabilityFields = {
...(typeof metadata.capabilities.vision === "boolean"
? { vision: metadata.capabilities.vision }
@@ -502,7 +512,9 @@ export function enrichCatalogModelEntry(
// #6241: surface thinking support + the canonical effort tiers so the frontend can
// render the effort/thinking toggles. `thinking` is kept for back-compat; `supportsThinking`
// is the explicit flag and `effort_tiers` lists the selectable reasoning levels
- // (only when the model actually supports thinking).
+ // (only when the model actually supports thinking). An explicit empty registry list
+ // is authoritative; GLM models also require a provider-declared contract instead of
+ // inheriting generic OpenAI effort tiers.
...(typeof metadata.capabilities.supportsThinking === "boolean"
? {
thinking: metadata.capabilities.supportsThinking,
diff --git a/src/lib/proxyLogger.ts b/src/lib/proxyLogger.ts
index a9eb4b3805d..8665e20c758 100644
--- a/src/lib/proxyLogger.ts
+++ b/src/lib/proxyLogger.ts
@@ -194,43 +194,103 @@ export function logProxyEvent(entry: ProxyLogInput) {
proxyLogs.length = MAX_IN_MEMORY_ENTRIES;
}
- // 2. Persist to SQLite
+ // 2. Queue for background batch persistence (SQLite / Redis)
if (shouldPersistToDisk) {
+ enqueueProxyLog(log);
+ }
+
+ return log;
+}
+
+// ──────────────── Background Batch Persistence ────────────────
+
+const BATCH_FLUSH_INTERVAL_MS = 1000;
+const BATCH_SIZE_THRESHOLD = 100;
+
+let pendingLogsQueue: ProxyLogEntry[] = [];
+let batchTimer: NodeJS.Timeout | null = null;
+
+function ensureBatchTimer() {
+ if (batchTimer) return;
+ batchTimer = setInterval(() => {
+ flushProxyLogsSync();
+ }, BATCH_FLUSH_INTERVAL_MS);
+ if (typeof batchTimer.unref === "function") {
+ batchTimer.unref();
+ }
+}
+
+function enqueueProxyLog(log: ProxyLogEntry) {
+ pendingLogsQueue.push(log);
+ ensureBatchTimer();
+ if (pendingLogsQueue.length >= BATCH_SIZE_THRESHOLD) {
+ flushProxyLogsSync();
+ }
+}
+
+export function flushProxyLogsSync() {
+ if (pendingLogsQueue.length === 0) return;
+ const batch = pendingLogsQueue;
+ pendingLogsQueue = [];
+
+ // 1. If Redis driver is active, asynchronously publish batch to Redis Stream/Channel
+ if (process.env.QUOTA_STORE_DRIVER === "redis" || process.env.QUOTA_STORE_REDIS_URL) {
try {
- const db = getDbInstance();
- db.prepare(
- `INSERT INTO proxy_logs (id, timestamp, status, proxy_type, proxy_host, proxy_port,
- level, level_id, provider, target_url, public_ip, egress_ip, latency_ms, error,
- connection_id, combo_id, account, tls_fingerprint)
- VALUES (@id, @timestamp, @status, @proxyType, @proxyHost, @proxyPort,
- @level, @levelId, @provider, @targetUrl, @clientIp, @egressIp, @latencyMs, @error,
- @connectionId, @comboId, @account, @tlsFingerprint)`
- ).run({
- id: log.id,
- timestamp: log.timestamp,
- status: log.status,
- proxyType: log.proxy?.type || null,
- proxyHost: log.proxy?.host || null,
- proxyPort: log.proxy?.port ? Number(log.proxy.port) : null,
- level: log.level,
- levelId: log.levelId,
- provider: log.provider,
- targetUrl: log.targetUrl,
- clientIp: log.clientIp,
- egressIp: log.egressIp,
- latencyMs: log.latencyMs,
- error: log.error,
- connectionId: log.connectionId,
- comboId: log.comboId,
- account: log.account,
- tlsFingerprint: log.tlsFingerprint ? 1 : 0,
- });
- } catch (err: any) {
- console.warn("[proxyLogger] Failed to persist:", err.message);
+ import("@/lib/quota/redisQuotaStore").then(({ getRedisQuotaStore }) => {
+ const store = getRedisQuotaStore(process.env.QUOTA_STORE_REDIS_URL || "");
+ const client = (store as any)?.client;
+ if (client && typeof client.publish === "function") {
+ for (const entry of batch) {
+ client.publish("omniroute:proxy_logs", JSON.stringify(entry)).catch(() => {});
+ }
+ }
+ }).catch(() => {});
+ } catch {
+ /* ignore redis pub errors */
}
}
- return log;
+ // 2. Persist to SQLite using a single transaction for high-performance non-blocking write
+ try {
+ const db = getDbInstance();
+ const insertStmt = db.prepare(
+ `INSERT INTO proxy_logs (id, timestamp, status, proxy_type, proxy_host, proxy_port,
+ level, level_id, provider, target_url, public_ip, egress_ip, latency_ms, error,
+ connection_id, combo_id, account, tls_fingerprint)
+ VALUES (@id, @timestamp, @status, @proxyType, @proxyHost, @proxyPort,
+ @level, @levelId, @provider, @targetUrl, @clientIp, @egressIp, @latencyMs, @error,
+ @connectionId, @comboId, @account, @tlsFingerprint)`
+ );
+
+ const transaction = db.transaction((entries: ProxyLogEntry[]) => {
+ for (const item of entries) {
+ insertStmt.run({
+ id: item.id,
+ timestamp: item.timestamp,
+ status: item.status,
+ proxyType: item.proxy?.type || null,
+ proxyHost: item.proxy?.host || null,
+ proxyPort: item.proxy?.port ? Number(item.proxy.port) : null,
+ level: item.level,
+ levelId: item.levelId,
+ provider: item.provider,
+ targetUrl: item.targetUrl,
+ clientIp: item.clientIp,
+ egressIp: item.egressIp,
+ latencyMs: item.latencyMs,
+ error: item.error,
+ connectionId: item.connectionId,
+ comboId: item.comboId,
+ account: item.account,
+ tlsFingerprint: item.tlsFingerprint ? 1 : 0,
+ });
+ }
+ });
+
+ transaction(batch);
+ } catch (err: any) {
+ console.warn("[proxyLogger] Failed to write proxy log batch to disk:", err?.message || err);
+ }
}
// ──────────────── Query ────────────────
diff --git a/src/lib/services/portProbe.ts b/src/lib/services/portProbe.ts
index 2a9fd502bda..9a890f541b4 100644
--- a/src/lib/services/portProbe.ts
+++ b/src/lib/services/portProbe.ts
@@ -168,11 +168,24 @@ export function parseSsPid(stdout: string): number | null {
export function parseNetstatPid(stdout: string, port: number): number | null {
for (const line of stdout.split("\n")) {
const columns = line.trim().split(/\s+/);
- // proto recv-q send-q local-address foreign-address state pid/program
+ // Linux: proto recv-q send-q local-address foreign-address state pid/program
if (columns.length < 7 || columns[5] !== "LISTEN") continue;
- if (!columns[3].endsWith(`:${port}`)) continue;
- const parsed = Number.parseInt(columns[6], 10);
- if (Number.isFinite(parsed)) return parsed;
+ const linuxAddress = columns[3].endsWith(`:${port}`);
+ const macAddress = columns[3].endsWith(`.${port}`);
+ if (!linuxAddress && !macAddress) continue;
+
+ if (linuxAddress) {
+ const linuxPid = Number.parseInt(columns[6], 10);
+ if (Number.isFinite(linuxPid)) return linuxPid;
+ }
+
+ // macOS `netstat -anv -p tcp` appends a `process:pid` column after
+ // the socket counters. Process names may contain spaces, so scan instead
+ // of relying on one fixed column index.
+ for (const column of columns.slice(6)) {
+ const match = /:(\d+)$/.exec(column);
+ if (match) return Number.parseInt(match[1], 10);
+ }
}
return null;
}
@@ -197,7 +210,11 @@ const PID_PROBES: ReadonlyArray<{
args: (port) => ["-tlnp", `sport = :${port}`],
parse: (stdout) => parseSsPid(stdout),
},
- { command: "netstat", args: () => ["-tlnp"], parse: parseNetstatPid },
+ {
+ command: "netstat",
+ args: () => (process.platform === "darwin" ? ["-anv", "-p", "tcp"] : ["-tlnp"]),
+ parse: parseNetstatPid,
+ },
];
/** Run one probe, resolving null on a missing binary, a non-match or a timeout. */
diff --git a/src/lib/usage/flatRateProviders.ts b/src/lib/usage/flatRateProviders.ts
index 8d3eec2c5d7..3454b9b3917 100644
--- a/src/lib/usage/flatRateProviders.ts
+++ b/src/lib/usage/flatRateProviders.ts
@@ -46,6 +46,11 @@ const FLAT_RATE_SUBSCRIPTION_PROVIDER_IDS: ReadonlySet = new Set([
"glm-cn", // GLM Coding (China) plan
"claude", // Claude Code plan (OAuth-only — a Claude Pro/Max subscription)
"cc", // Claude Code plan (alias id — same connection, shares the `cc` pricing rows)
+ // OpenCode Go subscription (https://opencode.ai/go) — a flat monthly fee. It is an
+ // aggregator reselling GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x, so
+ // per-token rows price each call at the UNDERLYING model's metered rate and the
+ // analytics overstatement is large rather than marginal (#11149).
+ "opencode-go",
]);
/**
diff --git a/src/lib/usage/internalUsageCommand.ts b/src/lib/usage/internalUsageCommand.ts
index 257b8f9c158..37f5f009d1c 100644
--- a/src/lib/usage/internalUsageCommand.ts
+++ b/src/lib/usage/internalUsageCommand.ts
@@ -13,7 +13,7 @@ const TEXT_PLAIN_HEADERS = { "Content-Type": "text/plain; charset=utf-8" } as co
type JsonRecord = Record;
-interface UsageCommandApiKeyMetadata {
+export interface UsageCommandApiKeyMetadata {
id: string;
name?: string;
allowedConnections?: string[] | null;
@@ -31,7 +31,7 @@ interface ProviderConnectionLike {
quotaWindowThresholds?: Record | null;
}
-interface UsageSnapshot {
+export interface UsageSnapshot {
connectionId: string;
provider: string;
plan: unknown;
@@ -39,7 +39,7 @@ interface UsageSnapshot {
quotaWindowThresholds?: Record | null;
}
-interface UsageCommandSelection {
+export interface UsageCommandSelection {
preferredProvider?: string | null;
preferredConnectionId?: string | null;
}
@@ -258,7 +258,7 @@ function snapshotFromConnection(
};
}
-async function collectUsageSnapshots(
+export async function collectUsageSnapshots(
metadata: UsageCommandApiKeyMetadata,
deps: RequiredDeps
): Promise {
@@ -525,6 +525,55 @@ function appendQuotaBlock(
lines.push(`⏱ reset in ${formatResetIn(getResetAt(match?.quota ?? null), now)}`);
}
+/**
+ * Structured form of the usage command — what {@link buildUsageCommandText}
+ * renders as text, exposed as data for API consumers (the OmniCopilot panel
+ * asks for it via `?format=json`). Text and JSON share the exact same
+ * collectors, so the two can never disagree about a number.
+ *
+ * The key design constraint is the 403 case: a key without `allowUsageCommand`
+ * must reach the client as a *structured* reason, not a bare text error — a
+ * caller rendering a usage panel has to be able to tell "the server does not
+ * know your limits yet" apart from "this key may not ask".
+ */
+/** Discriminated so the caller never reads a data field off a refusal:
+ * `allowed:false` carries only `error`; `allowed:true` carries the data. */
+export type UsageCommandJson =
+ | { allowed: false; error: { message: string } }
+ | {
+ allowed: true;
+ /** Present only when the key opted into per-key usage limits. */
+ personal: unknown | null;
+ /** The selected provider snapshot, or null when nothing is cached. */
+ provider: UsageSnapshot | null;
+ /** Every connection's snapshot, so a panel can render Codex / Claude /
+ * OpenCode side by side instead of only the selected one (#11191). The
+ * single-pick in `provider` is a presentation choice for a terminal; the
+ * collector already gathered all of them. */
+ providers: UsageSnapshot[];
+ };
+
+export async function buildUsageCommandJson(
+ metadata: UsageCommandApiKeyMetadata,
+ deps: InternalUsageCommandDeps = {},
+ selection: UsageCommandSelection = {}
+): Promise {
+ const resolvedDeps = await normalizeDeps(deps);
+ const personal =
+ metadata.usageLimitEnabled === true
+ ? await resolvedDeps.getApiKeyUsageLimitStatus(
+ {
+ ...metadata,
+ preferredProvider: selection.preferredProvider ?? metadata.preferredProvider ?? null,
+ },
+ { now: resolvedDeps.now }
+ )
+ : null;
+ const snapshots = await collectUsageSnapshots(metadata, resolvedDeps);
+ const provider = selectUsageSnapshot(snapshots, selection);
+ return { allowed: true, personal, provider, providers: snapshots };
+}
+
export async function buildUsageCommandText(
metadata: UsageCommandApiKeyMetadata,
deps: InternalUsageCommandDeps = {},
@@ -588,6 +637,17 @@ function inferHttpUsageCommandSelection(request: Request): UsageCommandSelection
}
}
+/** `?format=json` (or `?format=JSON`) — anything else falls back to the text
+ * form, which is the historical contract of this endpoint. */
+function wantsUsageCommandJson(request: Request): boolean {
+ try {
+ const format = new URL(request.url, "http://localhost").searchParams.get("format");
+ return format !== null && format.trim().toLowerCase() === "json";
+ } catch {
+ return false;
+ }
+}
+
function createPlainUsageCommandResponse(text: string, status = 200): Response {
return new Response(text, { status, headers: TEXT_PLAIN_HEADERS });
}
@@ -764,22 +824,45 @@ export async function handleInternalUsageCommandHttpRequest(
): Promise {
try {
const resolvedDeps = await normalizeDeps(deps);
+ const json = wantsUsageCommandJson(request);
const apiKey = extractUsageCommandApiKey(request);
if (!apiKey || !(await resolvedDeps.isValidApiKey(apiKey))) {
+ if (json) {
+ return Response.json(
+ { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson,
+ { status: 401 }
+ );
+ }
return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401);
}
const metadata = await resolvedDeps.getApiKeyMetadata(apiKey);
if (!metadata?.id) {
+ if (json) {
+ return Response.json(
+ { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson,
+ { status: 401 }
+ );
+ }
return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401);
}
if (metadata.allowUsageCommand !== true) {
+ if (json) {
+ return Response.json(
+ { allowed: false, error: { message: USAGE_COMMAND_DISABLED_MESSAGE } } satisfies UsageCommandJson,
+ { status: 403 }
+ );
+ }
return createPlainUsageCommandResponse(USAGE_COMMAND_DISABLED_MESSAGE, 403);
}
+ const selection = inferHttpUsageCommandSelection(request);
+ if (json) {
+ return Response.json(await buildUsageCommandJson(metadata, resolvedDeps, selection));
+ }
return createPlainUsageCommandResponse(
- await buildUsageCommandText(metadata, resolvedDeps, inferHttpUsageCommandSelection(request))
+ await buildUsageCommandText(metadata, resolvedDeps, selection)
);
} catch (err) {
const body = buildErrorBody(500, err instanceof Error ? err.message : String(err));
diff --git a/src/server/authz/pipeline.ts b/src/server/authz/pipeline.ts
index 9f4e46bed1e..d8dacf376d7 100644
--- a/src/server/authz/pipeline.ts
+++ b/src/server/authz/pipeline.ts
@@ -205,6 +205,7 @@ function drainingResponse(requestId: string): NextResponse {
{ status: 503 }
);
response.headers.set(AUTHZ_HEADER_REQUEST_ID, requestId);
+ response.headers.set("Retry-After", "5");
return response;
}
diff --git a/src/shared/components/ProviderIcon.tsx b/src/shared/components/ProviderIcon.tsx
index aeb6f777ba0..56d54bf0dc2 100644
--- a/src/shared/components/ProviderIcon.tsx
+++ b/src/shared/components/ProviderIcon.tsx
@@ -263,6 +263,7 @@ const KNOWN_PNGS = new Set([
"linkup-search",
"llamafile",
"llamagate",
+ "logfare",
"maritalk",
"nanobot",
"nanogpt",
diff --git a/src/shared/constants/providers.ts b/src/shared/constants/providers.ts
index 5c058958e76..b9b2b6951db 100644
--- a/src/shared/constants/providers.ts
+++ b/src/shared/constants/providers.ts
@@ -141,6 +141,7 @@ export const AGGREGATOR_PROVIDER_IDS = new Set([
"void-ai",
"helixmind",
"tabitoken",
+ "logfare",
]);
export const ENTERPRISE_CLOUD_PROVIDER_IDS = new Set([
diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts
index 2baf13d3aae..6dab614425e 100644
--- a/src/shared/constants/providers/apikey/gateways.ts
+++ b/src/shared/constants/providers/apikey/gateways.ts
@@ -673,11 +673,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
color: "#F97316",
textIcon: "G4F",
website: "https://g4f.space",
- hasFree: true,
- freeNote: "Free no-key reverse proxy to Groq (gpt4free project) — rate-limited to 5 req/min.",
+ hasFree: false,
+ freeNote:
+ "No-key reverse proxy to Groq (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.",
passthroughModels: true,
authHint:
- "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.",
+ "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.",
},
"g4f-gemini": {
id: "g4f-gemini",
@@ -687,11 +688,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
color: "#F97316",
textIcon: "G4F",
website: "https://g4f.space",
- hasFree: true,
- freeNote: "Free no-key reverse proxy to Gemini (gpt4free project) — rate-limited to 5 req/min.",
+ hasFree: false,
+ freeNote:
+ "No-key reverse proxy to Gemini (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.",
passthroughModels: true,
authHint:
- "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.",
+ "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.",
},
"g4f-pollinations": {
id: "g4f-pollinations",
@@ -701,12 +703,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
color: "#F97316",
textIcon: "G4F",
website: "https://g4f.space",
- hasFree: true,
+ hasFree: false,
freeNote:
- "Free no-key reverse proxy to Pollinations (gpt4free project) — rate-limited to 5 req/min.",
+ "No-key reverse proxy to Pollinations (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.",
passthroughModels: true,
authHint:
- "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.",
+ "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.",
},
"g4f-ollama": {
id: "g4f-ollama",
@@ -716,11 +718,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
color: "#F97316",
textIcon: "G4F",
website: "https://g4f.space",
- hasFree: true,
- freeNote: "Free no-key hosted Ollama gateway (gpt4free project) — rate-limited to 5 req/min.",
+ hasFree: false,
+ freeNote:
+ "No-key hosted Ollama gateway (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.",
passthroughModels: true,
authHint:
- "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.",
+ "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.",
},
"g4f-nvidia": {
id: "g4f-nvidia",
@@ -730,12 +733,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
color: "#F97316",
textIcon: "G4F",
website: "https://g4f.space",
- hasFree: true,
+ hasFree: false,
freeNote:
- "Free no-key reverse proxy to NVIDIA NIM (gpt4free project) — rate-limited to 5 req/min.",
+ "No-key reverse proxy to NVIDIA NIM (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.",
passthroughModels: true,
authHint:
- "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.",
+ "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.",
},
"vercel-ai-gateway": {
id: "vercel-ai-gateway",
@@ -1265,6 +1268,29 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
apiHint:
"Create a helix- key and use https://helixmind.online/v1. OpenAI requests use Bearer authentication; the Anthropic-compatible messages endpoint accepts x-api-key.",
},
+ // Logfare (https://logfare.ai) — free OpenAI-compatible inference, live-verified
+ // 2026-08-21 (real /v1/models catalog; 11 chat-capable models incl. kimi-k3,
+ // deepseek-v4-pro, glm-5.2, gpt-5.6-luna). Key issued instantly at /register
+ // (username/password, no email). ⚠️ Logfare logs every request in exchange for
+ // free inference (opt out at /consent) — surfaced in freeNote per the catalog
+ // convention for data-collecting free providers.
+ logfare: {
+ id: "logfare",
+ alias: "logfare",
+ name: "Logfare",
+ icon: "auto_awesome",
+ color: "#22C55E",
+ textIcon: "LF",
+ website: "https://logfare.ai",
+ hasFree: true,
+ freeNote:
+ "Free OpenAI-compatible inference — no rate limits, no card. Logfare logs every request (prompts, completions, metadata) for internal research; opt out at /consent. Read https://logfare.ai/tos and https://logfare.ai/privacy before use.",
+ authHint:
+ "Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token.",
+ apiHint:
+ "Create a free API key at https://logfare.ai/register, then use https://logfare.ai/v1 as the OpenAI-compatible base URL. Note the request-logging policy: prompts, completions and metadata are logged for research (opt out at https://logfare.ai/consent).",
+ passthroughModels: true,
+ },
// TabiToken (https://tabitoken.com) — NewAPI-based Claude gateway. Its public pricing
// endpoint lists a Claude-only catalog (Opus 5 / 4.8, each with a -thinking variant),
// every model accepting the Anthropic and OpenAI protocols.
diff --git a/src/shared/middleware/chatBodyAdmission.ts b/src/shared/middleware/chatBodyAdmission.ts
index 1c3c25b904b..9b20d78e5ae 100644
--- a/src/shared/middleware/chatBodyAdmission.ts
+++ b/src/shared/middleware/chatBodyAdmission.ts
@@ -18,6 +18,7 @@
import { CORS_HEADERS } from "../utils/cors";
import { createHmac } from "crypto";
import v8 from "node:v8";
+import { trackRequest } from "../../lib/gracefulShutdown";
function parsePositiveInt(value: string | undefined, fallback: number): number {
const parsed = Number.parseInt(String(value), 10);
@@ -229,6 +230,7 @@ export class ChatAdmissionController {
tryAcquireHealthyHeadroom(): ChatAdmissionLease | null {
if (this.#activeHealthy >= this.healthyHeadroom) return null;
this.#activeHealthy += 1;
+ const done = trackRequest();
let released = false;
return {
get released() {
@@ -238,6 +240,7 @@ export class ChatAdmissionController {
if (released) return;
released = true;
this.#activeHealthy = Math.max(0, this.#activeHealthy - 1);
+ done();
},
};
}
@@ -264,6 +267,7 @@ export class ChatAdmissionController {
tryAcquireHeavy(): ChatAdmissionLease | null {
if (this.#activeHeavy >= this.maxHeavyInFlight) return null;
this.#activeHeavy += 1;
+ const done = trackRequest();
let released = false;
return {
get released() {
@@ -273,6 +277,7 @@ export class ChatAdmissionController {
if (released) return;
released = true;
this.#activeHeavy = Math.max(0, this.#activeHeavy - 1);
+ done();
this.#dispatchFair();
},
};
diff --git a/src/shared/validation/schemas/combo.ts b/src/shared/validation/schemas/combo.ts
index edeca47e729..db825e11cc9 100644
--- a/src/shared/validation/schemas/combo.ts
+++ b/src/shared/validation/schemas/combo.ts
@@ -321,7 +321,7 @@ export const createComboSchema = z
.object({
name: comboNameSchema,
description: z.string().max(2000).optional(),
- models: z.array(comboModelEntry).optional().default([]),
+ models: z.array(comboModelEntry).min(1, "a combo requires at least one model"),
strategy: comboStrategySchema.optional().default("priority"),
config: comboRuntimeConfigSchema.optional(),
allowedProviders: z.array(z.string().trim().min(1).max(200)).max(100).optional(),
@@ -380,8 +380,9 @@ export const updateComboSchema = z
.object({
name: comboNameSchema.optional(),
description: z.string().max(2000).optional().nullable(),
- // Creation may leave `models` empty (`omniroute combo create` drafts one
- // that way); an update may not, or a working combo loses every target.
+ // An update may not remove every model from a combo, or a working combo
+ // loses every target. Creation refuses an empty list too: since the CLI
+ // gained --models (#10954), an empty draft has no remaining legitimate path.
models: z
.array(comboModelEntry)
.min(1, "an update cannot remove every model from a combo")
diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts
index 3ec6e6c6d6c..aadf7d29d3d 100644
--- a/src/sse/handlers/chat.ts
+++ b/src/sse/handlers/chat.ts
@@ -1851,7 +1851,7 @@ async function handleSingleModelChat(
modelPinned: runtimeOptions?.modelPinned ?? false,
routingComboId: runtimeOptions?.routingComboId ?? null,
sessionAffinityKey: runtimeOptions.sessionAffinityKey ?? null,
- reasoningTransportFallback: runtimeOptions.reasoningTransportFallback ?? "skip",
+ reasoningTransportFallback: runtimeOptions.reasoningTransportFallback ?? "drop",
managedLease: runtimeOptions.managedLease ?? null,
},
runtimeOptions
@@ -2240,6 +2240,7 @@ async function handleSingleModelChat(
if (
!runtimeOptions.emergencyFallbackTried &&
!comboName &&
+ !forceLiveComboTest &&
shouldRetrySameAccountTransport({
status: result.status,
errorText: errorStr,
diff --git a/src/sse/handlers/chatHelpers.ts b/src/sse/handlers/chatHelpers.ts
index 36bd4077408..bcf9a51c62c 100644
--- a/src/sse/handlers/chatHelpers.ts
+++ b/src/sse/handlers/chatHelpers.ts
@@ -422,7 +422,7 @@ export async function executeChatWithBreaker({
conversationId = null,
modelPinned = false,
routingComboId = null,
- reasoningTransportFallback = "skip",
+ reasoningTransportFallback = "drop",
sessionAffinityKey = null,
managedLease = null,
}: ExecuteChatWithBreakerOptions): Promise {
diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts
index b1df29552a4..5d3d8851786 100644
--- a/src/sse/services/auth.ts
+++ b/src/sse/services/auth.ts
@@ -106,8 +106,17 @@ import {
resolveProviderId,
NOAUTH_PROVIDERS,
WEB_COOKIE_PROVIDERS,
+ isSelfHostedChatProvider,
} from "@/shared/constants/providers";
-import { isModelExcludedByConnection } from "@/domain/connectionModelRules";
+import {
+ isModelExcludedByConnection,
+ isModelAdvertisedByConnection,
+} from "@/domain/connectionModelRules";
+import {
+ getSyncedAvailableModelsByConnection,
+ SYNCED_AVAILABLE_MODELS_MALFORMED,
+ type SyncedAvailableModelsByConnection,
+} from "@/lib/db/models";
import { isFreeModel } from "@/shared/utils/freeModels";
import {
applySessionAffinityPin,
@@ -1160,6 +1169,54 @@ function materializeConnection(
};
}
+/**
+ * #11089: load the per-connection synced model inventory for self-hosted chat
+ * providers so connection selection can drop hosts that never advertised the
+ * requested model.
+ *
+ * Scoped to SELF_HOSTED_CHAT_PROVIDER_IDS: those are the providers where one
+ * provider id fans out to several independent hosts with genuinely different
+ * inventories. Hosted providers share one catalog per provider, so filtering
+ * there would only add a DB read.
+ *
+ * Returns an empty map (= no filtering) when there is no model to match, when
+ * no candidate is self-hosted, or when the persisted rows are malformed — a
+ * partial read must never silently shrink the pool.
+ */
+async function loadAdvertisedModelsForSelfHostedConnections(
+ connections: ProviderConnectionView[],
+ requestedModel: string | null
+): Promise