diff --git a/AGENTS.md b/AGENTS.md index f10fb39483c..d468f7d1984 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -46,7 +46,7 @@ Repository map and Reference Documentation sections below. ## Project at a Glance -**OmniRoute** — unified AI proxy/router. One endpoint, 348 LLM providers, auto-fallback. +**OmniRoute** — unified AI proxy/router. One endpoint, 349 LLM providers, auto-fallback. | Layer | Location | Purpose | | ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | diff --git a/PROVIDER_REFERENCE.md b/PROVIDER_REFERENCE.md new file mode 100644 index 00000000000..571fe0e904c --- /dev/null +++ b/PROVIDER_REFERENCE.md @@ -0,0 +1,447 @@ +--- +title: "Provider Reference" +version: 3.8.50 +lastUpdated: 2026-08-21 +--- + +# Provider Reference + +> **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. +> Regenerate with: `npm run gen:provider-reference` +> **Last generated:** 2026-08-21 + +Total providers: **349**. See category breakdown below. + +## Categories + +- **Free** — free tier with API key (configured via dashboard) +- **No-auth** — public endpoints that require no key or sign-in at all +- **OAuth** — sign-in flow handled by OmniRoute, no API key needed +- **Web cookie** — wraps the provider's web app via cookie auth +- **API key** — paid provider configured via API key (free credits may apply) +- **Local** — runs on the user's machine (Ollama, LM Studio, vLLM, etc.) +- **Search** — web search providers +- **Audio** — audio-only providers (TTS/STT) +- **Upstream proxy** — providers that proxy to other providers +- **Cloud agent** — long-running coding agents (Codex Cloud, Devin, Jules) +- **System** — OmniRoute-internal providers (loopback, etc.) + +Additional tags: `image`, `video`, `aggregator`, `enterprise`, `embed/rerank`, `self-hosted`. + +`Tool calling` (where shown): `native` — real function-calling API; `emulated` — the `tools` array is prompt-emulated via `webTools.ts` (regex-parsed `{...}` blocks); `none` — `tools` is currently silently dropped. See #7286. + +Use the dashboard at `/dashboard/providers` to enable, configure, and test each provider. + +--- + +## No-auth Providers (no key required) (11) + +| ID | Alias | Name | Tags | Website | Notes | Tool calling | +|----|-------|------|------|---------|-------|--------------| +| `aihorde` | `horde` | AI Horde | No-auth | [link](https://aihorde.net) | No API key required — uses AI Horde's documented anonymous key. Adding a free aihorde.net key is optional and only buys higher queue priority (kudos). | — | +| `auggie` | `aug` | Augment (Auggie CLI) | No-auth | [link](https://augmentcode.com) | No API key stored by OmniRoute. Install the Auggie CLI and run `auggie login` on this machine, then OmniRoute spawns it locally for each request. | — | +| `chipotle` | `pepper` | Chipotle Pepper AI (Free) | No-auth | [link](https://amelia.chipotle.com) | No credentials required. Uses Chipotle's public support chatbot via reverse-engineered SockJS/STOMP protocol. | — | +| `cloudflare-playground` | `cfp` | Cloudflare AI Playground | No-auth | [link](https://playground.ai.cloudflare.com) | No credentials required — anonymous browser sessions over a reverse-engineered cf_agent WebSocket protocol (Playwright transport). | — | +| `devin-cli-agentic` | `dva` | Devin CLI Agentic Bridge | No-auth | [link](https://docs.devin.ai/work-with-devin/devin-cli) | Authentication is owned by the official Devin CLI in its isolated bridge volume. | emulated | +| `duckduckgo-web` | `ddgw` | DuckDuckGo AI Chat | No-auth | [link](https://duckduckgo.com/duckchat) | No credentials required — DuckDuckGo AI Chat is anonymous and free. | emulated | +| `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — | +| `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — | +| `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — | +| `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — | +| `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — | + +## OAuth Providers (25) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | +| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | +| `antigravity` | — | Antigravity | OAuth | — | — | +| `claude` | `cc` | Claude Code | OAuth | — | — | +| `cline` | `cl` | Cline | OAuth | — | — | +| `clinepass` | `cp` | ClinePass | OAuth | [link](https://cline.bot/cline-pass) | ClinePass is Cline's $9.99/mo subscription bundling 10 open coding models. Sign in with your Cline account (same login as the Cline CLI/IDE), or paste a direct ClinePass API key (app.cline.bot → Settings → API Keys). A ClinePass subscription unlocks the cline-pass/* models. Reuses the Cline WorkOS OAuth flow. | +| `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. | +| `codex` | `cx` | OpenAI Codex | OAuth | — | — | +| `cursor` | `cu` | Cursor IDE | OAuth | — | — | +| `devin-cli` | `dv` | Devin CLI | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | +| `devin-desktop` | — | Devin Desktop | OAuth | [link](https://devin.ai) | Paste an existing Devin API key from an authenticated Devin session. Key export availability and steps vary by Devin version and account. | +| `ghe-copilot` | `ghe-copilot` | GitHub Enterprise Copilot | OAuth | — | Enter your GHE instance URL (e.g., https://ghe.company.com) in provider settings, then authenticate via device flow. | +| `github` | `gh` | GitHub Copilot | OAuth | — | — | +| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab Duo OAuth is not configured. Register an OAuth application at https://gitlab.com/-/profile/applications with redirect URI http://localhost:20128/callback and scopes "ai_features read_user", then set GITLAB_DUO_OAUTH_CLIENT_ID (and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET) and restart. | +| `grok-cli` | `gc` | Grok Build | OAuth | — | Sign in with your browser, or paste your ~/.grok/auth.json (or the JWT access token) from the Grok Build CLI; refresh_token is rotated automatically either way. | +| `kilocode` | `kc` | Kilo Code | OAuth | — | — | +| `kimi-coding` | `kmc` | Kimi Code CLI | OAuth | [link](https://www.kimi.com/code?aff=omniroute) | Sign in with the same Kimi account used by Kimi Code CLI. OmniRoute uses the CLI OAuth flow and Kimi Coding Plan endpoints. | +| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | +| `openference` | `of` | Openference | OAuth | [link](https://openference.com) | Sign in with your Openference account to route requests through api.openference.com. An active plan is required for inference — OAuth may authenticate but return 402 without one. | +| `qoder` | `if` | Qoder | OAuth | — | — | +| `raycast` | `rc` | Raycast Pro AI | OAuth | [link](https://raycast.com/ai) | Unofficial integration — uses your Raycast Pro subscription via credentials from the macOS app (Auto-Import or manual capture). May break on Raycast updates. Not for redistribution; personal use only. | +| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | +| `xai-oauth` | `xao` | xAI OAuth (Grok) | OAuth | [link](https://x.ai) | Sign in with xAI to use api.x.ai models such as Grok 4.5. This is separate from Grok Build JWT sessions, which use cli-chat-proxy.grok.com and grok-build model aliases. | +| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | +| `zed-hosted` | — | Zed Hosted Models | OAuth | [link](https://zed.dev) | Sign in with your Zed account (native-app sign-in). OmniRoute generates a one-time RSA keypair and opens zed.dev to authorize it — on a remote/headless install, copy the resulting 127.0.0.1 callback URL from your browser's address bar and paste it back here. Distinct from the 'Zed IDE' credential-import entry above: this proxies chat completions through Zed's own hosted model aggregator (cloud.zed.dev), fronting Anthropic/OpenAI/Google/xAI models under your Zed plan. | + +## Web Cookie Providers (35) + +| ID | Alias | Name | Tags | Website | Notes | Tool calling | +|----|-------|------|------|---------|-------|--------------| +| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | emulated | +| `adobe-firefly` | `firefly` | Adobe Firefly (Image/Video) | Web cookie | [link](https://firefly.adobe.com) | RECOMMENDED: firefly.adobe.com signed-in → F12 → Network → click firefly-3p.ff.adobe.io (generate-async or models/discovery) → Request Headers → Authorization → copy the token AFTER 'Bearer ' (starts with eyJ…). Cookie-only from firefly.adobe.com mints a GUEST token → 401/403; only multi-domain IMS cookies (adobelogin.com) or that Bearer JWT work. Unofficial/experimental media + Limits. | — | +| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | emulated | +| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | emulated | +| `chatgpt-web-codex` | `cgpt-codex` | ChatGPT Web (Codex) | Web cookie | [link](https://chatgpt.com) | Paste the full ChatGPT Cookie header. OmniRoute verifies it in an isolated headless browser profile. | native | +| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | none | +| `conol-web` | `cnl` | Conol (Unofficial/Experimental) | Web cookie | [link](https://conol.ai) | Use browser sign-in, or paste the full Cookie header from conol.ai. The __Secure-better-auth.session_token cookie is required. | — | +| `copilot-m365-web` | `m365copilot` | Microsoft 365 Copilot (BizChat) | Web cookie | [link](https://m365.cloud.microsoft/chat) | Sign in at m365.cloud.microsoft/chat, then open DevTools → Network → filter 'WS' → click the Chathub WebSocket connection. Copy both the access_token query parameter AND the account-specific Chathub path segment from its request URL (wss://…/Chathub/?…&access_token=…). It is NOT an Authorization: Bearer header on an XHR/Fetch request. The token is short-lived; this is an unofficial integration. Optional: store a refresh_token in providerSpecificData.refreshToken (any Microsoft device-code/refresh flow for the substrate.office.com/sydney scopes) and OmniRoute pre-flight-refreshes the access token itself — otherwise re-capture after every ~75 min expiry. | — | +| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste the access_token from an authenticated copilot.microsoft.com request (DevTools → Network → Authorization), or export a HAR while logged in | — | +| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | emulated | +| `doubao-web` | `db` | Dola Web (ByteDance) | Web cookie | [link](https://www.dola.com) | Paste the full Cookie header from www.dola.com. It should include sessionid, ttwid, and s_v_web_id. If s_v_web_id is unavailable, fp=verify_... from a chat/completion request URL can be used as a fallback. | — | +| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — | +| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated | +| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — | +| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | +| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — | +| `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — | +| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated | +| `kimi-web` | `kimi-web` | Kimi Web | Web cookie | [link](https://www.kimi.com/code?aff=omniroute) | Paste access_token from www.kimi.com DevTools → Application → Local Storage. A legacy kimi-auth cookie is also accepted. | — | +| `lmarena` | `lma` | Arena (Free) | Web cookie | [link](https://arena.ai) | Paste the full Cookie header from arena.ai (DevTools → Network → request → Cookie). Include arena-auth-prod-v1.0/.1… and cf_clearance/__cf_bm when present. OmniRoute uses Chrome TLS impersonation; if Arena still 403s, set providerSpecificData.recaptchaV3Token from a live browser session. | — | +| `microsoft-designer-web` | `msdesigner` | Microsoft Designer (Image Generation) | Web cookie | [link](https://designer.microsoft.com) | Sign in at designer.microsoft.com, then open DevTools → Network, generate an image, and find the request to DallE.ashx?action=GetDallEImagesCogSci. Copy the value of its Authorization: Bearer header (the access_token — no 'Bearer ' prefix). The token is short-lived; this is an unofficial, reverse-engineered integration. | — | +| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your ecto_1_sess cookie AND the ecto1:... WS auth token from meta.ai. Capture the ecto1: token in DevTools → Network → WS → the clippy request's Authorization query param. Example: ecto_1_sess=4240a308...NVDg0; ecto1:ABCD... | emulated | +| `notion-web` | `nw` | Notion AI Web (Unofficial/Experimental) | Web cookie | [link](https://www.notion.so) | Paste only the token_v2 cookie VALUE from app.notion.com (DevTools → Application → Cookies → token_v2). Do not paste token_v2= or the full Cookie header. Workspace is auto-detected; space_id / notion_user_id are optional. | — | +| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | emulated | +| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | — | +| `promptql` | `pql` | PromptQL (Unofficial/Experimental) | Web cookie | [link](https://prompt.ql.app) | Paste the Bearer JWT from prompt.ql.app DevTools → Network → graphql → Authorization (token only). Optional projectId + session Cookie for refresh. | — | +| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | emulated | +| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | emulated | +| `tencent-aistudio-web` | `tasw` | Tencent AI Studio (Free) | Web cookie | [link](https://aistudio.tencent.ai) | Log in to aistudio.tencent.ai, open DevTools -> Network, copy any request Cookie header containing session tokens. | — | +| `tinycms-web` | `tcw` | TinyCMS Web (Free/Sub) | Web cookie | [link](https://site.tinycms.xyz) | Go to site.tinycms.xyz, open DevTools → Application → Local Storage, copy the value of 'app-config-uuid' (starts with 'R'), and paste it here. | — | +| `v0-vercel-web` | `v0-vercel-web` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | — | +| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | — | +| `yuanbao-web` | `ybw` | Tencent Yuanbao (Free) | Web cookie | [link](https://yuanbao.tencent.com) | Log in to yuanbao.tencent.com, then paste the full Cookie header (DevTools → Network → any /api request → Request Headers → Cookie). It must contain hy_user and hy_token. | — | +| `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | +| `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | + +## API Key Providers (paid / paid-with-free-credits) (233) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | +| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | +| `agnes` | `agnes` | Agnes AI | API key, video | [link](https://agnes-ai.com) | Get API key at agnes-ai.com | +| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | +| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | Free tier paused (2026) — AI/ML API is now pay-as-you-go only (min $20 top-up); no recurring free credits. | +| `ainative` | `ainative` | AINative Studio | API key | [link](https://ainative.studio) | Create a free API key at ainative.studio (no card), then paste it here as a Bearer token. | +| `aion` | `aion` | Aion Labs | API key | [link](https://www.aionlabs.ai) | Create a free API key at aionlabs.ai (no card), then paste it here as a Bearer token. | +| `alibaba` | `ali` | Alibaba Cloud Model Studio | API key | [link](https://bailian.console.alibabacloud.com/) | — | +| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.console.aliyun.com/) | — | +| `ant-ling` | `ling` | Ant Ling / Ring (inclusionAI) | API key | [link](https://developer.ant-ling.com/en/docs/) | Register and create an API key at the Ant Ling API console (https://chat.ant-ling.com/open), then paste it here. OmniRoute routes chat traffic to https://api.ant-ling.com/v1/chat/completions; the provider is OpenAI-compatible and also exposes an Anthropic-compatible surface. | +| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | +| `anyapi` | `anyapi` | AnyAPI AI | API key, aggregator | [link](https://anyapi.ai) | Free plan: 100,000 ANY Tokens/day and 100 RPM for eligible Free/Basic models; no credit card required. | +| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | +| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | +| `auriko` | `auriko` | Auriko | API key, aggregator | [link](https://www.auriko.ai) | Free plan publishes 1,000 Platform RPM and 10,000 BYOK RPM. Platform inference still passes through provider cost; this is not a free-token pool or unlimited free inference. | +| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | +| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | +| `bai` | `bai` | b.ai | API key | [link](https://b.ai) | Bearer API key for the b.ai OpenAI-compatible LLM gateway (distinct from TheB.AI). Create a key at https://docs.b.ai, then use https://api.b.ai/v1 as the OpenAI-compatible base URL. | +| `baichuan` | `baichuan` | Baichuan | API key | [link](https://www.baichuan-ai.com/) | Get API key at platform.baichuan-ai.com | +| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://ernie.baidu.com/) | Get API key at console.bce.baidu.com | +| `bailian-coding-plan` | `bcp` | Alibaba Token Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/token-plan-overview) | — | +| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | +| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | +| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | +| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply | +| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | +| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | +| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | +| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | +| `charm-hyper` | `charm-hyper` | Charm Hyper | API key | [link](https://hyper.charm.land) | 100 free monthly Hypercredits on signup | +| `chat-oripe` | `chat-oripe` | Chat Oripe | API key, aggregator | [link](https://api.oriper.com) | Official metadata advertises 2M tokens/month, but the public site and documentation were blocked during audit; treat the quota and brand mapping as unconfirmed. | +| `chatanywhere` | `chatanywhere` | ChatAnywhere | API key, aggregator | [link](https://chatanywhere.tech) | Personal, educational or research use only: public documentation cites 10,000 points/day and 200 requests/day per IP/key; do not use for commercial traffic. | +| `cheaperinference` | `cinf` | Cheaper Inference | API key | [link](https://cheaperinference.com/?utm_source=omniroute) | — | +| `chenzk` | `chenzk` | Chenzk API | API key | [link](https://chenzk.top) | — | +| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | +| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | +| `cloudcode-one` | `cloudcode-one` | CloudCode.ONE | API key, aggregator | [link](https://cloudcode.one) | Published free models include glm-4.7-flash and glm-4.6v-flash; no numeric quota is published, and key creation may require credit or a coupon. | +| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | +| `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — | +| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | +| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | +| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | +| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | +| `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. | +| `dahl` | `dahl` | Dahl | API key | [link](https://inference.dahl.global) | Click 'Add Account' to auto-generate a token, or add a manual API key. | +| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | +| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | +| `deepai` | `deepai` | DeepAI | API key, image | [link](https://deepai.org) | Use your DeepAI API key. Get one at deepai.org — requires a Pro subscription ($9.99/mo). | +| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | +| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | +| `dgrid` | `dgrid` | DGrid | API key | [link](https://dgrid.ai) | DGrid Free Models Router: 10 requests/minute and 100 requests/day. A $5 lifetime top-up unlocks up to 20 requests/minute and 1,000 requests/day. | +| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | +| `digitalocean` | `digitalocean` | DigitalOcean | API key | [link](https://docs.digitalocean.com/products/ai-platform/) | — | +| `dit` | `dai` | DIT.ai | API key | [link](https://dit.ai) | Use your dit.ai API key in Authorization: Bearer . Fully OpenAI-compatible — a drop-in replacement, just change the base URL to https://api.dit.ai/v1. | +| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | +| `dxnt` | `dxnt` | DXNT / DX Token | API key, aggregator | [link](https://www.dxnt.com) | Free accounts are documented at 100 calls/day; the quota may increase through invitations and can vary by account. | +| `electronhub` | `electronhub` | Electron Hub | API key, aggregator | [link](https://www.electronhub.ai) | Free plan: 5 RPM, $0.25 weekly credits and 10 Neutrinos/day for :free models; family budgets also apply. | +| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | +| `factory` | `factory` | Factory | API key | [link](https://factory.ai) | Bearer API key for the Factory OpenAI-compatible gateway. | +| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | +| `fastrouter` | `fastrouter` | FastRouter | API key, aggregator | [link](https://fastrouter.ai) | Models with the :free suffix allow 10 requests/day per organization and model; availability may change. | +| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | +| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | +| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | +| `free-ai` | `free-ai` | Free.ai | API key, aggregator | [link](https://free.ai) | 30,000 tokens/day cover self-hosted models after email verification. Usage beyond the pool can bill at raw cost, and premium external models are paid. | +| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | +| `freebuff` | `freebuff` | Freebuff | API key | [link](https://freebuff.com) | Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester). | +| `freeinference` | `freeinference` | FreeInference | API key, aggregator | [link](https://freeinference.org) | Free research access without a card; non-Harvard applicants require manual approval and no numeric quota is publicly guaranteed. | +| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | +| `freetheai` | `fta` | FreeTheAi | API key, aggregator | [link](https://freetheai.xyz) | Join the FreeTheAi Discord to get your free API key. | +| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | +| `g4f-gemini` | `g4fgem` | g4f.space — Gemini | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-groq` | `g4fgroq` | g4f.space — Groq | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-nvidia` | `g4fnv` | g4f.space — NVIDIA | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-ollama` | `g4foll` | g4f.space — Ollama | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-pollinations` | `g4fpol` | g4f.space — Pollinations | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | ⚠️ **DEPRECATED.** api.galadriel.ai no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | +| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free tier available through Google AI Studio; current per-model quotas and regional limits apply | +| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | +| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | +| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | +| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free MiMo (xiaomi/mimo-v2.5) revoked 2026-05 — Opengateway is now a pay-as-you-go credit gateway; no recurring free model. | +| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free Nemotron promo ended 2026-06 — the GMI Cloud route is now pay-as-you-go credit only. | +| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | +| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | +| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | +| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | +| `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | +| `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | +| `helyxai` | `helyxai` | Helyx AI | API key, aggregator | [link](https://helyxai.space) | Operational Free plan documents 100,000 tokens/day; the site's separate 2M+ marketing claim conflicts and is not treated as a quota guarantee. | +| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | +| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | +| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | +| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | +| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `inception` | `inception` | Inception | API key | [link](https://docs.inceptionlabs.ai) | 10M free tokens on signup, no credit card required. | +| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | +| `internlm` | `internlm` | InternLM (Intern-S1) | API key | [link](https://internlm.intern-ai.org.cn/) | Free monthly quota ~1M input / 3M output tokens (~10 RPM) | +| `jina-ai` | `jina` | Jina AI (Foundation API) | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for api.jina.ai — embeddings, rerank, classify, segment, and search. Dashboard keys take precedence over JINA_AI_API_KEY. This is not the Reader / r.jina.ai card and does not fetch URLs. | +| `jina-reader` | `jr` | Jina Reader (r.jina.ai) | API key | [link](https://jina.ai/reader) | Bearer API key for r.jina.ai URL-to-markdown (/v1/web/fetch only). Does not serve /v1/embeddings or /v1/rerank. The same Jina token as Foundation API works; OmniRoute reuses a jina-ai dashboard key or JINA_AI_API_KEY when this card is empty. | +| `kenari` | `kenari` | Kenari | API key | [link](https://kenari.id) | Use your Kenari API key (kn-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://kenari.id/v1. | +| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | +| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | +| `kimi` | `kimi` | Kimi (Legacy Moonshot API) | API key | [link](https://platform.kimi.ai?aff=omniroute) | — | +| `kimi-coding-apikey` | `kmca` | Kimi Code API Key | API key | [link](https://www.kimi.com/code?aff=omniroute) | — | +| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | +| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | +| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | +| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | +| `literouter` | `literouter` | LiteRouter | API key, aggregator | [link](https://literouter.com) | Free model variants use the :free suffix; daily credit limits vary by model and free input is capped at 5,000 tokens. | +| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | +| `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. | +| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | +| `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | +| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. | +| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | +| `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | +| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | +| `meganova-ai` | `meganova-ai` | MegaNova AI | API key, aggregator | [link](https://meganova.ai) | Free signup without a card. Published Tier 1 per-model quotas total 550 requests/day; they are not a shared global pool, and paid overage can apply if enabled. | +| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | +| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | +| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | +| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | +| `mixedbread` | `mxbai` | Mixedbread AI | API key | [link](https://www.mixedbread.com) | Bearer API key for the Mixedbread embeddings API. | +| `mixlayer` | `mixlayer` | Mixlayer | API key, aggregator | [link](https://www.mixlayer.com) | The qwen/qwen3.5-4b-free model is free for prototyping and rate-limited; no fixed public RPM or daily quota is confirmed. | +| `mnn-ai` | `mnn-ai` | MNN AI | API key, aggregator | [link](https://mnnai.ru) | Free plan: $1 monthly credits, 10 RPM and access only to models marked Free. | +| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | +| `modelscope` | `ms` | ModelScope | API key | [link](https://modelscope.cn) | Free tier via ModelScope API-Inference — Alibaba account required. | +| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | ⚠️ **DEPRECATED.** Monster API shuttered operations on 2026-06-30. Use alternative OpenAI-compatible providers. | +| `moonshot` | `moonshot` | Kimi | API key | [link](https://platform.kimi.ai?aff=omniroute) | — | +| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | +| `muse-code` | `mc` | Muse Code (Meta) | API key | [link](https://github.com/meta-llama/llama-stack) | Use your META_API_KEY env var as a Bearer token. Muse Code CLI uses the OpenAI Responses API wire format (POST /responses). | +| `naga-ac` | `naga` | Naga.ac | API key, aggregator | [link](https://naga.ac) | Get API key at naga.ac — Google/GitHub/Discord signup available. | +| `naga-ai` | `naga-ai` | Naga AI | API key, aggregator | [link](https://naga.ac) | Models marked :free are publicly listed, but no numeric quota is confirmed. Naga's policy warns that free-tier prompts and outputs may be collected or used for training. | +| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | +| `nara` | `nara` | NaraRouter | API key | [link](https://bynara.id) | Get a free API key via NaraRouter's Telegram channel, then paste it here as a Bearer token. | +| `navy` | `navy` | NavyAI | API key | [link](https://api.navy) | Create a free API key from the NavyAI dashboard, then paste it here as a Bearer token. | +| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | +| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | +| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | +| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | +| `novita` | `novita` | Novita AI | API key, video, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | +| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | +| `nube` | `nube` | Nube.sh | API key | [link](https://nube.sh) | — | +| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | +| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | +| `ofoxai` | `ofoxai` | OfoxAI | API key, aggregator | [link](https://ofox.ai) | The current catalog advertises 10+ free models without a public numeric quota; review upstream provenance, retention and training terms before production use. | +| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/keys) | — | +| `openadapter` | `oad` | OpenAdapter | API key | [link](https://openadapter.dev) | Use your OpenAdapter API key in Authorization: Bearer sk-cv-. Fully OpenAI-compatible. API base URL: https://api.openadapter.in/v1. | +| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | +| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | +| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | +| `openference-api` | `ofa` | Openference API | API key | [link](https://openference.com) | Free plan: 3-day trial with open-source models — no credit card required | +| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | +| `openvecta` | `openvecta` | OpenVecta | API key | [link](https://openvecta.com) | Free credits on signup for OpenAI-compatible inference across LLMs, embeddings, and reasoning models | +| `orcarouter` | `orcarouter` | OrcaRouter | API key | [link](https://www.orcarouter.ai) | — | +| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | +| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | +| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | +| `pioneer` | `pn` | Pioneer AI | API key | [link](https://pioneer.ai) | $75 free usage credits — no credit card required | +| `plamo` | `plamo` | PLaMo | API key | [link](https://plamo.preferredai.jp/api) | — | +| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | +| `poixe-ai` | `poixe-ai` | Poixe AI | API key, aggregator | [link](https://poixe.com) | Current public free limits are small and model-group specific: 2 RPM/5 RPD for large-cup models and 20 RPM/50 RPD for small-cup models. | +| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | Anonymous/keyless access to the documented free models is best-effort. Local v3.8.50 verification (2026-07-31) returned 401 via OmniRoute and Cloudflare 1010 on direct upstream probes from the same network. Premium models still require a Pollinations API key from enter.pollinations.ai. | +| `poolside` | `poolside` | Poolside | API key | [link](https://poolside.ai) | Laguna S 2.1 and XS 2.1 are free during Preview; no public numeric quota is published. | +| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | ⚠️ **DEPRECATED.** serving.app.predibase.com no longer resolves (sweep 2026-06-19); the managed serving API appears discontinued. | +| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid | +| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product-s/qianfan_home) | — | +| `qiniu` | `qiniu` | Qiniu | API key | [link](https://www.qiniu.com) | — | +| `qwen-cloud` | `qwc` | Qwen Cloud | API key | [link](https://www.qwencloud.com/) | — | +| `qwen-cloud-token-plan` | `qct` | Qwen Cloud Token Plan | API key | [link](https://www.qwencloud.com/pricing/token-plan) | — | +| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | +| `regolo` | `regolo` | Regolo AI | API key | [link](https://regolo.ai) | Get your Regolo API key from regolo.ai, then paste it here as a Bearer token. | +| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | +| `requesty` | `requesty` | Requesty | API key | [link](https://requesty.ai) | Free tier ~200 requests/day - multi-model routing gateway (300+ models) | +| `routeway` | `routeway` | Routeway | API key | [link](https://routeway.ai) | Create a free API key at routeway.ai, then paste it here as a Bearer token. | +| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | +| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | +| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | +| `sarvam` | `sarvam` | Sarvam AI | API key | [link](https://docs.sarvam.ai) | ₹1,000 in free signup credits — never expire | +| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/docs/ai-data/generative-apis/) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | +| `sealion` | `sealion` | SEA-LION | API key | [link](https://sea-lion.ai) | Sign in at sea-lion.ai with Google (no card, no region wall), create an API key, then paste it here. | +| `segmind` | `segmind` | Segmind | API key, image, video | [link](https://segmind.com) | Use your Segmind API key in the x-api-key header. OmniRoute targets https://api.segmind.com/v1/ and returns the generated image/video bytes directly. | +| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | +| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus currently listed $0 models after identity verification; availability and limits may change | +| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | +| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `speka` | `speka` | Speka AI | API key, aggregator | [link](https://speka.me) | Free plan: $1 monthly usage, 10 RPM, one API key and access to open models and the playground; no card required. | +| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | +| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | +| `sumopod` | `sumopod` | SumoPod | API key | [link](https://ai.sumopod.com) | Use your SumoPod API key (sk-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://ai.sumopod.com/v1. | +| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | +| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | +| `tabitoken` | `tabitoken` | TabiToken | API key, aggregator | [link](https://tabitoken.com) | — | +| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | +| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | +| `tinyfish` | `tf` | TinyFish Fetch | API key | [link](https://docs.tinyfish.ai/fetch-api) | X-API-Key from agent.tinyfish.ai/api-keys | +| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | — | +| `token-kiosk` | `tk` | Token Kiosk | API key | [link](https://agent-router.gaib.ai) | Use your Token Kiosk API key in Authorization: Bearer . Fully OpenAI-compatible gateway. API base URL: https://agent-router.gaib.ai/v1. | +| `tokenreply` | `tokenreply` | TokenReply | API key, aggregator | [link](https://www.tokenreply.com) | Free-tagged models have model- and campaign-specific daily limits; no fixed global free quota is published. | +| `tokenrouter` | `trk` | TokenRouter | API key | [link](https://tokenrouter.com) | Use your TokenRouter API key in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.tokenrouter.com/v1. | +| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | +| `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. | +| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | +| `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. | +| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | +| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | +| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | +| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | +| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | +| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | +| `void-ai` | `void-ai` | Void AI | API key, aggregator | [link](https://voidai.app) | The public model catalog marks some models with a free plan requirement, but access is conditional and no numeric quota is confirmed. | +| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | +| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | +| `wafer` | `wafer` | Wafer AI | API key | [link](https://wafer.ai) | — | +| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | +| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | +| `writer` | `writer` | Writer | API key | [link](https://dev.writer.com) | — | +| `x5lab` | `x5lab` | X5Lab | API key | [link](https://x5lab.dev) | Use your X5Lab API key (x5-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.x5lab.dev/v1. | +| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | Use an official xAI API key, or sign in with xAI OAuth. Grok Build JWT sessions remain a separate provider. | +| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | +| `xiaomi-mimo-token-plan` | `mimotp` | Xiaomi MiMo Token Plan | API key | [link](https://mimo.mi.com) | — | +| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | +| `yolo-auto` | `yolo-auto` | Yolo-Auto | API key, aggregator | [link](https://yolo-auto.com) | Free API access is request-limited and intended for testing; no numeric daily quota is published and free access is not promised indefinitely. | +| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | +| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. | +| `zerolimitai` | `zerolimitai` | ZeroLimitAI | API key, aggregator | [link](https://www.zerolimitai.com) | Temporary free trial is advertised, but official pages conflict between 3 and 7 days; a 100-calls/day claim is not treated as permanent. | +| `zylo-api` | `zylo` | Zylo API | API key, aggregator | [link](https://zyloai.net) | Basic plan: 10 RPM, 7,200 requests/day and 200,000 tokens/day; limited to Basic text models. | + +## Local Providers (14) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | +| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | +| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | +| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | +| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | +| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | +| `mlx-gemma` | `mlx-gemma` | MLX Gemma 26B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11435. Requires uv and mlx-lm installed. Model: mlx-community/gemma-4-26B-A4B-it-qat-q4_0-mlx-aligned (~15.9GB peak memory). | +| `mlx-qwen` | `mlx-qwen` | MLX Qwen 3.8 27B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11436. Requires uv and mlx-lm installed. Model: maglun/Qwen3.8-27B-MLX-Mixed-3.80bpw (~13.1GB peak memory). | +| `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. | +| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | +| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | +| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | + +## Search Providers (13) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | +| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | +| `firecrawl` | `fc` | Firecrawl | Search | [link](https://firecrawl.dev) | API key from firecrawl.dev/app/api-keys (or set your self-hosted Firecrawl base URL) | +| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | +| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | +| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/keys) | Same API key as Ollama Cloud (from ollama.com/settings/keys) | +| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | +| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs/google) | API key from SearchAPI (query param or Bearer auth) | +| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | +| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | +| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | +| `x-search` | `x_search` | X Search (Grok) | Search | [link](https://docs.x.ai/developers/tools/x-search) | SuperGrok OAuth (xai-oauth) or xAI API key. This is Grok X Search, not the X Developer MCP. | +| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/business/api/) | X-API-Key from the You.com platform dashboard | + +## Audio-only Providers (12) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | +| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | +| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | +| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | +| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | +| `fishaudio` | `fishaudio` | Fish Audio | Audio | [link](https://fish.audio) | — | +| `gladia` | `gladia` | Gladia | Audio | [link](https://gladia.io) | — | +| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | +| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | +| `rev-ai` | `revai` | Rev AI | Audio | [link](https://www.rev.ai) | — | +| `soniox` | `sx` | Soniox | Audio | [link](https://soniox.com) | — | +| `speechmatics` | `sm` | Speechmatics | Audio | [link](https://www.speechmatics.com) | Free tier — 8 hours/month, no credit card required. Batch (async) mode only. | + +## Upstream Proxy Providers (2) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | +| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | + +## Cloud Agent Providers (3) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | +| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | +| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | + +## System Providers (1) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `auto` | `auto` | Auto (Zero-Config) | System | — | — | + +## Sources of truth + +- Catalog: [`src/shared/constants/providers.ts`](../../src/shared/constants/providers.ts) +- Registry (per-model details): [`open-sse/config/providerRegistry.ts`](../../open-sse/config/providerRegistry.ts) +- Executors: [`open-sse/executors/`](../../open-sse/executors/) (106 implementations) +- Translators: [`open-sse/translator/`](../../open-sse/translator/) + +## See Also + +- [FREE_TIERS.md](./FREE_TIERS.md) — curated free-tier guide +- [USER_GUIDE.md](../guides/USER_GUIDE.md) — provider setup walkthrough +- [ARCHITECTURE.md](../architecture/ARCHITECTURE.md) — overall architecture diff --git a/README.md b/README.md index 55e6d0328a1..6600b406492 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 348 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 348 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 349 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 349 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. @@ -101,7 +101,7 @@ ⚙️ Features 🎯 Combos - 🌐 Providers + 🌐 Providers 🔌 CLI & MCP @@ -210,7 +210,7 @@ curl http://localhost:20128/v1/chat/completions \ -The Promise — One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 348 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests). +The Promise — One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 349 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests).

@@ -461,7 +461,7 @@ All **19** strategies — mix & match per combo step: -What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 348 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. +What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 349 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. 📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md) @@ -559,7 +559,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute - **🖼️ New endpoints** — `/v1/ocr` (Mistral OCR) and `/v1/audio/translations` (Whisper-style) round out the media surface. → [API Reference](docs/reference/API_REFERENCE.md) - **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md) - **🌍 Deployment & ops** — reverse-proxy `basePath`, browser-language auto-detect, per-key device tracking, root-less MITM trust, zh-TW localization. → [Environment](docs/reference/ENVIRONMENT.md) -- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **348-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) +- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **349-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) - **📡 Routing transparency** — every response carries an `X-OmniRoute-Decision` header naming the strategy/provider/latency that served it, a new `cache-optimized` combo strategy + Auto-Combo `cacheAffinity` factor route repeat requests back to the connection holding the cached prefix, and a read-only `/v1/auto-combo/{channel}/candidates` endpoint exposes an `auto/*` channel's live candidate pool. → [Auto-Combo](docs/routing/AUTO-COMBO.md) - **⚡ Local performance & infra** — one-click local Redis, Cloudflare Workers / Deno Deploy relay deployers, Bifrost & Mux as supervised embedded services. → [Embedded Services](docs/frameworks/EMBEDDED-SERVICES.md) @@ -642,11 +642,11 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
-## 🌐 348 AI Providers — 90+ Free +## 🌐 349 AI Providers — 90+ Free
-> The most complete catalog of any open-source router: **348 providers**, **90+ with a free tier**, **56 free forever**. +> The most complete catalog of any open-source router: **349 providers**, **90+ with a free tier**, **56 free forever**.
@@ -990,11 +990,11 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ `:latest` follows the highest **published** stable SemVer. It does not track git `main`. Pin `:X.Y.Z` for GitOps. See [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels).The image pins **`OMNIROUTE_MEMORY_MB=1024`**. That is enough for the dashboard and a light chat. **Coding agents** (`POST /v1/responses` from Claude Code, Codex, Grok, …) need a much larger V8 heap or the process `FATAL ERROR`s at ~12 GiB under two overlapping long contexts. Size the container above the heap (native buffers sit outside V8): -| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | -| --- | --- | --- | -| Dashboard / light chat | `1024` (image default) | ≥2 g | -| One coding agent | `8192` | ≥10 g | -| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | +| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | +| ----------------------------------- | ------------------------------- | ---------------------- | +| Dashboard / light chat | `1024` (image default) | ≥2 g | +| One coding agent | `8192` | ≥10 g | +| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | ```bash docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ @@ -1003,6 +1003,7 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ ``` Full table: [Docker Guide — runtime RAM](docs/guides/DOCKER_GUIDE.md#runtime-ram-for-coding-agents). + > **Pre-release Docker channel:** `diegosouzapw/omniroute:next` and > `diegosouzapw/omniroute:next-web` follow the current default `release/v*` > branch. These mutable tags are intended only for testing unreleased fixes and diff --git a/bin/cli/commands/combo.mjs b/bin/cli/commands/combo.mjs index 554c1c1d838..8d58cf73bd9 100644 --- a/bin/cli/commands/combo.mjs +++ b/bin/cli/commands/combo.mjs @@ -307,6 +307,12 @@ export async function runComboCreateCommand(name, strategy = "priority", opts = } const models = Array.isArray(opts.models) ? opts.models : []; + if (!models.length) { + console.error( + "combo create requires at least one target. Pass --models and/or repeat --model ." + ); + return 1; + } try { return await withRuntime(async ({ kind, api, db }) => { diff --git a/changelog.d/features/10987-logfare-free-provider.md b/changelog.d/features/10987-logfare-free-provider.md new file mode 100644 index 00000000000..507a528411f --- /dev/null +++ b/changelog.d/features/10987-logfare-free-provider.md @@ -0,0 +1 @@ +- **feat(providers):** add Logfare as a free OpenAI-compatible provider — dashboard card with a Free badge and request-logging disclosure (every prompt/completion is logged for research; opt out at logfare.ai/consent), live model discovery from `https://logfare.ai/v1/models` (20 models, 11 chat-capable: kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3…), full chat/streaming through the existing OpenAI-compatible path, the real Logfare logo on the card, and a listing in the free-tiers guide. ([#10987](https://github.com/diegosouzapw/OmniRoute/pull/10987)) diff --git a/changelog.d/features/11190-usage-command-json.md b/changelog.d/features/11190-usage-command-json.md new file mode 100644 index 00000000000..d7655f04c51 --- /dev/null +++ b/changelog.d/features/11190-usage-command-json.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage` gains a structured form — `?format=json` returns the key's own usage as `ApiKeyUsageLimitStatus` + `UsageSnapshot` instead of `text/plain`. This is the surface a UI (the OmniCopilot panel) consumes to show a key holder their daily/weekly spend and quota reset. The route is self-service (the caller's own key, gated by `allowUsageCommand`), not the management surface; refusals come back as a discriminated `{ "allowed": false, "error": … }` so a UI can tell "not allowed" apart from "allowed but nothing cached yet". The endpoint was previously undocumented in `API_REFERENCE.md`; it now has a section ([#11190](https://github.com/diegosouzapw/OmniRoute/pull/11190)) diff --git a/changelog.d/features/11192-usage-command-providers-array.md b/changelog.d/features/11192-usage-command-providers-array.md new file mode 100644 index 00000000000..b7ef4211091 --- /dev/null +++ b/changelog.d/features/11192-usage-command-providers-array.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage?format=json` now returns `providers[]` — every connection's quota snapshot, not just the single selected one — so a panel can render Codex / Claude / OpenCode side by side. The collector already gathered all of them; the single-pick `provider` field (kept) is a terminal presentation choice. Closes the per-connection gap from OmniCopilot #8 ([#11192](https://github.com/diegosouzapw/OmniRoute/pull/11192)) diff --git a/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md b/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md new file mode 100644 index 00000000000..47f8de84fd2 --- /dev/null +++ b/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md @@ -0,0 +1 @@ +- **fix(providers):** the five g4f.space sub-providers (Groq, Gemini, Pollinations, Ollama, NVIDIA) no longer advertise a free tier — a keyless `POST /v1/chat/completions` now returns `402 insufficient_credits` behind a proof-of-work "cake" wall (re-verified live 2026-08-22), so `hasFree` is `false` and the notes point at `g4f.dev/members.html`. The gateway still works with a member key, so its registry wiring and `authType: "optional"` are unchanged ([#10071](https://github.com/diegosouzapw/OmniRoute/issues/10071)) — thanks @chirag127 diff --git a/changelog.d/fixes/10550-responses-reasoning-transport.md b/changelog.d/fixes/10550-responses-reasoning-transport.md index d34c433debd..e2b40cdb8c9 100644 --- a/changelog.d/fixes/10550-responses-reasoning-transport.md +++ b/changelog.d/fixes/10550-responses-reasoning-transport.md @@ -1 +1 @@ -- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Combos now drop incompatible continuation reasoning by default and can explicitly skip incompatible targets, while known providers no longer show redundant encrypted-reasoning controls. (#10550) +- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Direct requests drop incompatible continuation reasoning by default; combos can explicitly skip incompatible targets without mutating the request. Known providers no longer show redundant encrypted-reasoning controls. (#10550, #10959) diff --git a/changelog.d/fixes/10949-mixed-reasoning-plaintext.md b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md new file mode 100644 index 00000000000..05a055ec538 --- /dev/null +++ b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md @@ -0,0 +1 @@ +- Preserve explicit plaintext reasoning when a Responses reasoning item also carries opaque provider state (rare OpenCode Go `deepseek-v4-flash` responses). Mixed plaintext + opaque input is projected onto the target transport: plaintext targets keep portable text, opaque targets keep provider state. Opaque-only reasoning is dropped when the selected target cannot replay it, allowing cross-model conversations to continue. (#10949, #10959) diff --git a/changelog.d/fixes/11015-shutdown-track-sse.md b/changelog.d/fixes/11015-shutdown-track-sse.md new file mode 100644 index 00000000000..1ed99b3669d --- /dev/null +++ b/changelog.d/fixes/11015-shutdown-track-sse.md @@ -0,0 +1 @@ +- **fix(resilience):** count heavyweight `/v1` admission leases in the SIGTERM drain and send `Retry-After` on shutdown 503s so Recreate no longer looks like an empty 502 ([#11015](https://github.com/diegosouzapw/OmniRoute/issues/11015)) — thanks @RaviTharuma diff --git a/changelog.d/fixes/11089-chat-routing-synced-inventory.md b/changelog.d/fixes/11089-chat-routing-synced-inventory.md new file mode 100644 index 00000000000..922b96a659b --- /dev/null +++ b/changelog.d/fixes/11089-chat-routing-synced-inventory.md @@ -0,0 +1 @@ +- **fix(resilience):** filter chat connection selection by each connection's *synced* model inventory on multi-host self-hosted providers (`ollama-local`, `lm-studio`, `vllm`, …), so a request for a model only one host advertises is pinned to that host instead of failing over onto a host that never had it ([#11089](https://github.com/diegosouzapw/OmniRoute/issues/11089)) diff --git a/changelog.d/fixes/11149-opencode-go-flat-rate.md b/changelog.d/fixes/11149-opencode-go-flat-rate.md new file mode 100644 index 00000000000..7aa63ad4559 --- /dev/null +++ b/changelog.d/fixes/11149-opencode-go-flat-rate.md @@ -0,0 +1 @@ +- **fix(analytics):** `opencode-go` is now classified as a flat-rate subscription, so cost analytics shows $0 for it instead of billing every call at the underlying model’s metered rate — it resells GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x under one flat monthly fee, which made the overstatement large rather than marginal ([#11149](https://github.com/diegosouzapw/OmniRoute/pull/11149)) — thanks @electrumguy diff --git a/changelog.d/fixes/11162-combo-create-requires-model.md b/changelog.d/fixes/11162-combo-create-requires-model.md new file mode 100644 index 00000000000..228e6a9b20f --- /dev/null +++ b/changelog.d/fixes/11162-combo-create-requires-model.md @@ -0,0 +1 @@ +- **Combo create:** creating a routing combo without any model is now refused (`400`) — the CLI requires `--models`/`--model` on `combo create`, matching the dashboard which already rejected empty combos. diff --git a/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md b/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md new file mode 100644 index 00000000000..c533feae543 --- /dev/null +++ b/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md @@ -0,0 +1 @@ +- **fix(routing):** a custom `openai-compatible-*` / `anthropic-compatible-*` connection pointing at a keyless self-hosted backend (llama.cpp, Ollama, vLLM started without an API key) now stays in the `auto/*` candidate pool instead of being silently dropped by the credential gate — for those IDs "no credential" is the normal configuration, not an unconfigured connection ([#11180](https://github.com/diegosouzapw/OmniRoute/pull/11180)) — thanks @marcs7 diff --git a/changelog.d/fixes/11181-lkgp-enabled-context.md b/changelog.d/fixes/11181-lkgp-enabled-context.md new file mode 100644 index 00000000000..d1c0cde5a36 --- /dev/null +++ b/changelog.d/fixes/11181-lkgp-enabled-context.md @@ -0,0 +1 @@ +- **fix(routing):** the Routing tab's "last known good provider" toggle now actually takes effect — `lkgpEnabled` was persisted and the `lkgp` strategy guarded on it, but the setting was never forwarded into the `RoutingContext` built in `resolveAutoStrategyOrder()`, so `context.lkgpEnabled` was always `undefined` and the off-switch was unreachable ([#11181](https://github.com/diegosouzapw/OmniRoute/issues/11181)) diff --git a/changelog.d/fixes/9763-ratelimit-mintime-floor.md b/changelog.d/fixes/9763-ratelimit-mintime-floor.md new file mode 100644 index 00000000000..2b145f1e1a3 --- /dev/null +++ b/changelog.d/fixes/9763-ratelimit-mintime-floor.md @@ -0,0 +1 @@ +- **fix(ratelimit):** respect operator `minTimeBetweenRequestsMs` floor when relaxing the limiter on headroom — the adaptive rate-limit learning no longer silently erases a configured minimum gap between requests when the upstream reports plenty of remaining capacity ([#9763](https://github.com/diegosouzapw/OmniRoute/issues/9763)). diff --git a/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md new file mode 100644 index 00000000000..82a88c5905a --- /dev/null +++ b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md @@ -0,0 +1 @@ +- **fix(executors):** OpencodeExecutor rotates (or retries once on a single-account direct path) on upstream 400 empty-body rejections — malformed completion envelopes with no error field were propagated as success and killed client sessions. Bounded +1 attempt per request; body reads are conditioned on status 400 so successful/streaming responses are never buffered. 400s carrying an error field keep propagating immediately. diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index 3dae8591dd7..79875e31489 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -853,11 +853,6 @@ "count": 1 } }, - "src/app/api/usage/call-logs/route.ts": { - "no-restricted-imports": { - "count": 1 - } - }, "src/app/api/usage/quota/route.ts": { "no-restricted-imports": { "count": 1 @@ -953,11 +948,6 @@ "count": 1 } }, - "src/app/api/v1/rerank/route.ts": { - "no-restricted-imports": { - "count": 1 - } - }, "src/app/api/v1/vscode/[token]/models/route.ts": { "no-restricted-syntax": { "count": 1 @@ -3259,4 +3249,4 @@ "count": 5 } } -} +} \ No newline at end of file diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index b7f40cebfa6..0a3239ea7fb 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,5 +1,6 @@ { "_rebaseline_2026_08_20_10531_freebuff_provider": "PR #10531 (adrianaryaputra, feat/freebuff-provider-support, closes #6793) own growth: src/shared/constants/providers/apikey/gateways.ts 1283->1298 (+15, the freebuff APIKEY_PROVIDERS_GATEWAYS catalog entry, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines) and src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx 1062->1067 (+5, freebuff credential placeholder/hint at the existing per-provider switch chokepoint). Covered by tests/unit/freebuff-provider.test.ts (9/9 passing).", + "_rebaseline_2026_08_21_10987_logfare_provider": "PR #10987 (jonlwheat2-gif, feat/10644-logfare-provider, closes #10644) own growth: src/shared/constants/providers/apikey/gateways.ts 1298->1321 (+23, the logfare APIKEY_PROVIDERS_GATEWAYS catalog entry with Free badge/freeNote/apiHint documenting the request-logging policy, additive data at the existing registry chokepoint, same god-file no-split rationale as the prior gateways.ts rebaselines: #10531 freebuff, merge-storm 2026-08-11). Covered by tests/unit/logfare-registry.test.ts (1/1 passing).", "_rebaseline_2026_08_20_10574_reasoning_transport_fallback": "PR #10574 (jackjinke, fix/responses-reasoning-transport, fixes #10550) own growth: src/sse/handlers/chatHelpers.ts 1017->1019 (+2 = the new reasoningTransportFallback option threaded through executeChatWithBreaker's options destructure and its downstream handleSingleModel call, at the existing per-attempt options-passthrough chokepoint; not extractable without splitting the option-forwarding call itself). Covered by the PR's own reasoning-policy test suite (tests/unit/chatcore-translation-paths.test.ts, tests/unit/combo-attempt-body-isolation-7847.test.ts, tests/unit/reasoning-cache.test.ts, tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts among others), 446/446 focused tests passing.", "_rebaseline_2026_08_18_10517_zed_hosted_oauth_callback_port": "PR #10517 (phatchau036, fix/zed-hosted-oauth-callback-port) own growth: src/shared/components/OAuthModal.tsx 1131->1148 (wc -l; check-file-size.mjs counts via split(\"\\n\").length so the gate sees 1134->1149, +15/+18, crosses the frozen 1134 cap). Wires the zed-hosted native-app callback auto-complete: forceManual gating on isTrueLocalhost for zed-hosted, the loopback-redirect-URI comment block, and the exchangeToken full-URL-as-code branch, all at the existing provider-switch chokepoints this modal already carries growth for (seventh bump: 969->989->993->998->1030->1056->1100->1149; structural shrink tracked in #3501). The actual port-derivation logic lives in src/lib/oauth/providers/zed-hosted.ts (not frozen here) and was hardened during pre-merge review to use the server's own getRuntimePorts() instead of a browser-guessed scheme/port, covered by the new tests/unit/zed-hosted-loopback-port-derivation.test.ts (8/8 passing).", "_rebaseline_2026_08_13_10243_codex_fingerprint_merge": "PR #10243 (xz-dev, Codex OAuth fingerprint convergence) merge into release/v3.8.50: src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts crossed the 1000-line new-file cap for the first time (974 on base, 997 on the PR's own branch, 1013 after merging + prettier reflow) purely from combining two independent, already-legitimate feature additions that landed on the same shared UI-helper file — this PR's own Codex fingerprint-mode select/toggle wiring (CODEX_FINGERPRINT_MODE_VALUES, getCodexFingerprintModeLabel, CodexFingerprintModeValue) plus #8949's unrelated Codex account-service-tier helpers merged concurrently on release/v3.8.50. Neither addition alone crosses the cap; git's line-level auto-merge does not detect a threshold crossing. Not modularized as part of this conflict-resolution merge commit (out of scope — this is a merge, not a feature change). Covered by the PR's own tests/unit/codex-fingerprint-convergence.test.ts, tests/unit/executor-codex.test.ts, tests/unit/provider-specific-data-schema.test.ts (all passing post-merge).", @@ -388,6 +389,10 @@ "open-sse/services/claudeCodeCompatible.ts": 1563, "open-sse/services/combo.ts": 4742, "open-sse/services/compression/strategySelector.ts": 1379, + "open-sse/services/compression/engines/ccr/index.ts": 1024, + "_rebaseline_2026_08_22_11084_ccr_caller_gate": "PR #11084 (HouMinXi) own growth: open-sse/services/compression/engines/ccr/index.ts 1000->1024 (first listing — the engine was unlisted and drifted just over the 1000 cap; +24 are the callerSupportsCcrRetrieve gate that skips replacement entirely for callers without the retrieve tool, closing the stranded-prompt incident measured in production). Covered by tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", + "open-sse/services/contextManager.ts": 1001, + "_rebaseline_2026_08_22_11113_purify_system_first": "PR #11113 (ggdayup) own growth: open-sse/services/contextManager.ts 1000->1001 (+1, purifyHistory merges the compression notice into the leading system message instead of splicing a second one mid-array — live-confirmed TokenRouter 400s; the +1 is the merge-into-leading branch, not extractable). Covered by tests/unit/context-manager-purify-system-first.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "open-sse/services/rateLimitManager.ts": 1517, "open-sse/translator/response/openai-responses.ts": 1652, "open-sse/utils/cursorAgentProtobuf.ts": 1956, @@ -427,7 +432,8 @@ "src/shared/components/analytics/charts.tsx": 1346, "src/shared/services/cliRuntime.ts": 1459, "src/sse/handlers/chat.ts": 2493, - "src/sse/services/auth.ts": 3260, + "src/sse/services/auth.ts": 3337, + "_rebaseline_2026_08_23_11186_synced_inventory_routing": "PR #11186 (pacocartones) own growth: src/sse/services/auth.ts 3260->3337 (+77, loadAdvertisedModelsForSelfHostedConnections + the modelNotAdvertised candidate-filter predicate — pins chat routing to the connection whose synced inventory actually advertises the model, fixing spurious model-not-found on multi-host self-hosted setups; at the existing credential-selection chokepoint, not extractable without splitting the selection flow). Covered by tests/unit/chat-routing-synced-inventory-11089.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "tests/unit/account-fallback-service.test.ts": 2044, "tests/unit/provider-validation-specialty.test.ts": 3880, "open-sse/executors/hyperagent.ts": 1334, @@ -436,7 +442,8 @@ "open-sse/executors/kiro.ts": 1390, "open-sse/translator/request/openai-to-kiro.ts": 1374, "open-sse/utils/sseHeartbeat.ts": 194, - "open-sse/utils/proxyFetch.ts": 1239, + "open-sse/utils/proxyFetch.ts": 1244, + "_rebaseline_2026_08_23_11177_dns_retry_classification": "PR #11177 (rqzbeh) own growth: proxyFetch.ts 1239->1244 (+5, EAI_AGAIN/ENOTFOUND/ETIMEDOUT join the retryable dispatcher classification alongside ECONNREFUSED — bounded socket retries for transient DNS failures, part of the #10443 Hermes→Antigravity stream-drop fixes). Covered by tests/unit/proxy-fetch-dns-retry-10443.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry: DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legítima acima do cap; gateways.ts = god-file de catálogo de providers que cresceu com os PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o próprio PR #9421 foi o que quebrou o arquivo; sem split até o release, congelado no tamanho atual). Owner autorizou rebaseline com anotação (2026-08-11).": { "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1062, "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051, @@ -447,7 +454,7 @@ "_rebaseline_2026_08_22_11156_enter_check_disabled": "PR #11156 (rqzbeh) own growth: AddApiKeyModal.tsx 1080->1082 (+2, Enter keydown handler now mirrors the isCheckDisabled condition — owner-requested post-merge polish from #11056; the rest of the diff is Prettier reflow). Covered by tests/unit/ui/add-api-key-modal-enter-key.test.tsx (jsdom render test, Enter dispatch assertions).", "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051, "src/shared/components/ModelSelectModal.tsx": 1138, - "src/shared/constants/providers/apikey/gateways.ts": 1298, + "src/shared/constants/providers/apikey/gateways.ts": 1321, "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1387, "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry": "DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legitima acima do cap; gateways.ts = god-file de catalogo de providers que cresceu com PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o proprio PR #9421 quebrou o arquivo); bridge.ts (PR #8949) = ponte Chromium vendored; proxyFetch.ts 1207->1220 = drift herdado de merges. Owner autorizou rebaseline com anotacao (2026-08-11).", "src/lib/modelCapabilities.ts": 1072, @@ -455,7 +462,8 @@ "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 1014, "open-sse/config/imageRegistry.ts": 1034, "src/sse/handlers/chatHelpers.ts": 1019, - "src/shared/middleware/chatBodyAdmission.ts": 1005, + "src/shared/middleware/chatBodyAdmission.ts": 1009, + "_rebaseline_2026_08_22_11020_sigterm_drain": "PR #11020 (RaviTharuma) own growth: chatBodyAdmission.ts 1005->1009 (+4, heavyweight admission leases now increment the SIGTERM drain counter and releaseChatAdmissionWhenDone holds it for the SSE lifetime — closes #11015; +4 are the lease/drain wiring lines at the existing admission chokepoint). Covered by tests/unit/chat-body-admission.test.ts heavyweight-lease cases. Owner pre-authorized baseline bumps 2026-08-22.", "_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file).", "open-sse/executors/commandCode.ts": 1059, "_rebaseline_2026_08_21_10859_vision_bridge_catalog": "#10859 own growth (Vision Bridge fixes #10808/#10809): src/lib/modelCapabilities.ts 1006->1016 (+10, cmd/gpt-5.3-codex* text-only capability resolution) and open-sse/executors/commandCode.ts 988->1023 (+35, Command Code wire-model normalization for bare ids + reasoning field fallback for opencode-routed gateways). Cohesive bug fixes at the existing capability-resolution / executor chokepoints; not extractable mid-fix. Covered by tests/unit/model-capabilities-command-code-codex-textonly-10703.test.ts, tests/unit/command-code-vision.test.ts, tests/unit/opencode-mimo-reasoning-details-nonstream.test.ts. Pushed directly to release (own-session miss: the original rebaseline was made in a throwaway validation worktree and never landed on the PR branch or the release before merge).", diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index 95e35a5adec..604a5a8e19d 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -197,10 +197,11 @@ "_rebaseline_2026_08_09_v3850_release_close": "7666 -> 8045 (+379 gzip bytes, +4.9%). Release v3.8.50 close reconciliation measured twice with the real size-limit + @size-limit/file path on tip e0ce95c592. Per-entry measurements remain below their absolute budgets: omniroute.mjs 4380/15000, mcp-server.mjs 1195/5000, nodeRuntimeSupport.mjs 887/8000, reset-password.mjs 1583/6000. The growth accumulated through legitimate CLI/runtime work in this cycle, including global-install ESM alias resolution, Termux cache preparation, and MCP stdio startup hardening; no entrypoint is near its absolute ceiling. The direction:down ratchet stays blocking from this exact measured tip." }, "openapiBreaking": { - "value": 0, + "value": 4, "direction": "down", "dedicatedGate": true, - "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec." + "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec.", + "_rebaseline_2026_08_22_combo_create_min1": "0 -> 4, split 3 own + 1 inherited. Docs-only alignment of components.schemas.ComboCreate with the request contract already enforced by the API since 638fc5fbd (combo create refuses an empty model list) and d5034ea52: `model`/`nodes` were phantom properties the server never accepted, and `models` (array, minItems 1) is the real required field. OWN findings (3, caused by this commit): removed `model`, removed `nodes`, added required `models` on POST /api/combos — spec-vs-server drift, not client-facing breakage, no working client could have relied on the removed shapes. INHERITED finding (1, NOT caused by this PR's code changes — pre-existing drift already present at parent d5034ea52): PATCH /api/combos/{id} request-body-added-required; that route's patch operation declares its own inline requestBody (required: true, bare object schema, docs/openapi.yaml ~2107-2118) and does not reference ComboCreate, so this finding exists independently of the ComboCreate alignment (same own-growth vs inherited-drift convention as _rebaseline_2026_07_20_aliasresolver_hook_split_7808). No code change in this PR; follow-up tracking = this change's PR description." }, "mutationScore.src/sse/services/auth.ts": { "value": 52.57, diff --git a/docs/changelog/fragments/10962.md b/docs/changelog/fragments/10962.md new file mode 100644 index 00000000000..5170a415e3f --- /dev/null +++ b/docs/changelog/fragments/10962.md @@ -0,0 +1 @@ +fix(catalog): expose only provider-routable GLM reasoning-effort tiers and remove unroutable ZCode aliases diff --git a/docs/diagrams/cli-terminal.svg b/docs/diagrams/cli-terminal.svg index e2ad57c8b13..6ec68793d0c 100644 --- a/docs/diagrams/cli-terminal.svg +++ b/docs/diagrams/cli-terminal.svg @@ -1,4 +1,4 @@ - + Compact animated terminal cycling three real OmniRoute CLI commands with a typewriter effect and a scrolling subcommand ticker; the first frame shows the completed providers-list screen. diff --git a/docs/diagrams/comparison-table.svg b/docs/diagrams/comparison-table.svg index 053194678cc..f61e7ca0add 100644 --- a/docs/diagrams/comparison-table.svg +++ b/docs/diagrams/comparison-table.svg @@ -1,4 +1,4 @@ - + Static-header comparison table where each capability row fades in top to bottom; the OmniRoute column is highlighted and shows a check or a leading value in every row, while competitors show a mix of checks, partials and crosses. diff --git a/docs/diagrams/promise-pillars.svg b/docs/diagrams/promise-pillars.svg index 99b7f36b15a..aebefefabbf 100644 --- a/docs/diagrams/promise-pillars.svg +++ b/docs/diagrams/promise-pillars.svg @@ -1,4 +1,4 @@ - + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. @@ -21,7 +21,7 @@ - One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. + One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. @@ -38,7 +38,7 @@ Never hit limits - Auto-fallback across 348 providers in + Auto-fallback across 349 providers in milliseconds. Quota out? The next provider takes over — zero downtime. diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index 99543d2471e..3448cddc7ff 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -28,7 +28,7 @@ Never stop coding. - Every AI tool → 348 providers — 90+ free — through one endpoint. + Every AI tool → 349 providers — 90+ free — through one endpoint. Claude Code · Codex · Cursor · Cline · Copilot · Antigravity  →  FREE Claude / GPT / Gemini · auto-fallback diff --git a/docs/getting-started/FREE-TIERS-GUIDE.md b/docs/getting-started/FREE-TIERS-GUIDE.md index 008fe962577..6fd9dcc35bd 100644 --- a/docs/getting-started/FREE-TIERS-GUIDE.md +++ b/docs/getting-started/FREE-TIERS-GUIDE.md @@ -26,6 +26,7 @@ These providers have a recurring, keyless, or uncapped free-access path in the a | **Kiro AI** | Claude Sonnet 4.5, Haiku 4.5, DeepSeek V3.2, and others | Audited catalog estimates a 25K-token shared monthly pool | OAuth/account flow; ToS flagged `avoid` in the catalog | | **OpenCode Free** | Current `*-free` model set in the provider registry | Keyless; no published token cap | No provider credential; ToS flagged `avoid` | | **Pollinations** | Current keyless model set; some former models are discontinued or key-required | Keyless; no published token cap | No provider credential for the keyless models | +| **Logfare** | kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3, and more | Free API key (no rate limits, no card); **every request is logged** for research (opt out at logfare.ai/consent) | Instant key at logfare.ai/register; ToS/privacy at logfare.ai/tos and logfare.ai/privacy | | **Cloudflare AI** | Workers AI catalog | Audited pool estimates ~30M tokens/month from published usage units | Cloudflare account and API credentials | | **Gemini** | Gemini Flash family | Audited pool estimates ~60M tokens/month | Google AI Studio API key; rate limits apply | | **Groq** | Llama, GPT-OSS, and Qwen models | Audited pool estimates ~15M tokens/month | Groq API key; rate limits apply | diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 61d414de272..b24e7b7f610 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -505,7 +505,7 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. | Constraint | Consequence | | --- | --- | | Single writer | Do **not** run multiple replicas against the same SQLite file. That corrupts the DB. | -| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. | +| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. New requests during the empty-endpoint window get a reverse-proxy **`502 Bad Gateway: Unknown error`**, not OmniRoute JSON — clients cannot distinguish this from a provider failure (#11015). | | Same event loop as `/healthz` | A busy catalog or compression tick can delay probes; a short timeout then restarts the **only** replica. | **Probe matrix** (see also [Kubernetes probe recommendations](../ops/MONITORING_GUIDE.md#kubernetes-probe-recommendations)): @@ -518,6 +518,35 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. **Upgrades:** expect every session to drop. Drain clients if you can; there is no rolling update on default SQLite. Compose `restart: unless-stopped` plus Docker `HEALTHCHECK` will also replace the only process when the container is Unhealthy — same blast radius. +Kubernetes snippet for a **single replica** (Recreate is required; do not raise `replicas` against one SQLite file): + +```yaml +spec: + replicas: 1 + strategy: + type: Recreate + template: + spec: + terminationGracePeriodSeconds: 90 + containers: + - name: omniroute + lifecycle: + preStop: + exec: + command: ["/bin/sleep", "15"] + readinessProbe: + httpGet: + path: /healthz + port: 20128 + periodSeconds: 5 + livenessProbe: + tcpSocket: + port: 20128 + periodSeconds: 20 +``` + +`preStop` sleep lets kube drop Service endpoints before SIGTERM so **new** traffic stops hitting the dying process. In-flight `/v1/responses` SSE is drained up to `SHUTDOWN_TIMEOUT_MS` (default 30s) via heavyweight admission leases (#11015). New requests that still reach the process get `503` + `Retry-After: 5`. The Recreate empty-endpoint gap until the replacement is Ready remains a hard outage — that is the SQLite topology, not a probe misconfig. + External Postgres / multi-writer HA is **not** a documented stock path. If you need HA, keep a single replica or run a topology the project has tested and documented separately. The Postgres/MySQL work lives in [#8075](https://github.com/diegosouzapw/OmniRoute/issues/8075). Until that ships, the only supported way to multiply **large** `/v1/responses` capacity is N independent processes (next section), not `replicas > 1` on one volume. ## Scale-out: N independent processes diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt index 188e4c546fd..23e919c6280 100644 --- a/docs/i18n/ar/llm.txt +++ b/docs/i18n/ar/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/az/llm.txt b/docs/i18n/az/llm.txt index 1778044af16..fcb2eb38907 100644 --- a/docs/i18n/az/llm.txt +++ b/docs/i18n/az/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt index 1778044af16..fcb2eb38907 100644 --- a/docs/i18n/bg/llm.txt +++ b/docs/i18n/bg/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/bn/llm.txt b/docs/i18n/bn/llm.txt index e1f037d6698..f15efad72be 100644 --- a/docs/i18n/bn/llm.txt +++ b/docs/i18n/bn/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt index c181079187f..5c4acb63f9a 100644 --- a/docs/i18n/cs/llm.txt +++ b/docs/i18n/cs/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt index fea53e17b2a..e4ca44f80ca 100644 --- a/docs/i18n/da/llm.txt +++ b/docs/i18n/da/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt index 494914668a9..833ca47bf6b 100644 --- a/docs/i18n/de/llm.txt +++ b/docs/i18n/de/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt index 6dfe5a95b36..332a3a47890 100644 --- a/docs/i18n/es/llm.txt +++ b/docs/i18n/es/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fa/llm.txt b/docs/i18n/fa/llm.txt index 55c6e845de3..c6de54f2711 100644 --- a/docs/i18n/fa/llm.txt +++ b/docs/i18n/fa/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt index ca8eb2a75b3..dda61789985 100644 --- a/docs/i18n/fi/llm.txt +++ b/docs/i18n/fi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt index 6fc229f45d5..4dd7139b7e0 100644 --- a/docs/i18n/fr/llm.txt +++ b/docs/i18n/fr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/gu/llm.txt b/docs/i18n/gu/llm.txt index 0ef4d516e71..ecb0741ac4f 100644 --- a/docs/i18n/gu/llm.txt +++ b/docs/i18n/gu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt index e3d3b77ab8e..de5e856bac4 100644 --- a/docs/i18n/he/llm.txt +++ b/docs/i18n/he/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt index f92757a9f3f..2a6d15b8697 100644 --- a/docs/i18n/hi/llm.txt +++ b/docs/i18n/hi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt index e3ba1fa0a65..078d59d88ee 100644 --- a/docs/i18n/hu/llm.txt +++ b/docs/i18n/hu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt index 5a169f9e862..ba42452fc49 100644 --- a/docs/i18n/id/llm.txt +++ b/docs/i18n/id/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/in/llm.txt b/docs/i18n/in/llm.txt index e0b18cb0baa..79c9a604265 100644 --- a/docs/i18n/in/llm.txt +++ b/docs/i18n/in/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt index dbde86f284b..12f0f6555a4 100644 --- a/docs/i18n/it/llm.txt +++ b/docs/i18n/it/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt index 4bae4264258..87ca4d8a177 100644 --- a/docs/i18n/ja/llm.txt +++ b/docs/i18n/ja/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt index a2310374798..a6203810bed 100644 --- a/docs/i18n/ko/llm.txt +++ b/docs/i18n/ko/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/mr/llm.txt b/docs/i18n/mr/llm.txt index d40eb377976..dc8c256fa97 100644 --- a/docs/i18n/mr/llm.txt +++ b/docs/i18n/mr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt index cdf457c441e..9efa530fb7d 100644 --- a/docs/i18n/ms/llm.txt +++ b/docs/i18n/ms/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt index b2539aceb9e..838311afb37 100644 --- a/docs/i18n/nl/llm.txt +++ b/docs/i18n/nl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt index 9b4dedc1274..2e7e4f30e4b 100644 --- a/docs/i18n/no/llm.txt +++ b/docs/i18n/no/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt index d9b674c08eb..f40f7ff96c7 100644 --- a/docs/i18n/phi/llm.txt +++ b/docs/i18n/phi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt index 4c43c03d360..b8e216bf410 100644 --- a/docs/i18n/pl/llm.txt +++ b/docs/i18n/pl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt index ecbeb9ad4c8..92402048cc2 100644 --- a/docs/i18n/pt-BR/llm.txt +++ b/docs/i18n/pt-BR/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt index 7cd77fdd4a6..1d8ea55c53b 100644 --- a/docs/i18n/pt/llm.txt +++ b/docs/i18n/pt/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt index a4f03e94e7a..cb8a15c28f7 100644 --- a/docs/i18n/ro/llm.txt +++ b/docs/i18n/ro/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt index 8fe9b5bdb68..1a1c527403d 100644 --- a/docs/i18n/ru/llm.txt +++ b/docs/i18n/ru/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt index c5a67d9242e..9ecbf58571b 100644 --- a/docs/i18n/sk/llm.txt +++ b/docs/i18n/sk/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt index 3b7b7b674ea..e3e56731c6b 100644 --- a/docs/i18n/sv/llm.txt +++ b/docs/i18n/sv/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sw/llm.txt b/docs/i18n/sw/llm.txt index 4c7bec4311b..9f19d4a92ef 100644 --- a/docs/i18n/sw/llm.txt +++ b/docs/i18n/sw/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ta/llm.txt b/docs/i18n/ta/llm.txt index 5323983d315..952ef1584c2 100644 --- a/docs/i18n/ta/llm.txt +++ b/docs/i18n/ta/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/te/llm.txt b/docs/i18n/te/llm.txt index ce196f99275..7ca0ad8bc91 100644 --- a/docs/i18n/te/llm.txt +++ b/docs/i18n/te/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt index 39eb9c04ff8..fa1c6d5ba90 100644 --- a/docs/i18n/th/llm.txt +++ b/docs/i18n/th/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt index 7f848b6cdd6..72ffb228ff9 100644 --- a/docs/i18n/tr/llm.txt +++ b/docs/i18n/tr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt index bca18382ad2..f2fe37d1f85 100644 --- a/docs/i18n/uk-UA/llm.txt +++ b/docs/i18n/uk-UA/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ur/llm.txt b/docs/i18n/ur/llm.txt index 85369bee2e9..e731f3ccf57 100644 --- a/docs/i18n/ur/llm.txt +++ b/docs/i18n/ur/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt index da6b9b7179a..38e56d30c5e 100644 --- a/docs/i18n/vi/llm.txt +++ b/docs/i18n/vi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt index 0c737e5dc43..b3ca89b4d17 100644 --- a/docs/i18n/zh-CN/llm.txt +++ b/docs/i18n/zh-CN/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/zh-TW/llm.txt b/docs/i18n/zh-TW/llm.txt index 643943547df..a9c38ef389e 100644 --- a/docs/i18n/zh-TW/llm.txt +++ b/docs/i18n/zh-TW/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/openapi.yaml b/docs/openapi.yaml index e51b6091aea..73941f37ead 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -8893,12 +8893,20 @@ components: ComboCreate: type: object - required: [name, model] + required: [name, models] properties: name: type: string - model: - type: string + models: + type: array + minItems: 1 + items: + oneOf: + - type: string + description: "provider/model reference" + - type: object + description: "structured combo step (provider, model, weight, ...)" + additionalProperties: true strategy: type: string enum: @@ -8920,14 +8928,3 @@ components: - context-optimized - fusion default: priority - nodes: - type: array - items: - type: object - properties: - connectionId: - type: string - weight: - type: integer - priority: - type: integer diff --git a/docs/reference/API_REFERENCE.md b/docs/reference/API_REFERENCE.md index 359ae4750ec..1b71551e373 100644 --- a/docs/reference/API_REFERENCE.md +++ b/docs/reference/API_REFERENCE.md @@ -636,6 +636,49 @@ completion. --- +## Self-service usage (`/api/usage/om-usage`) + +Any API key can read **its own** usage and quotas — no management auth. This is the endpoint a +client (CLI, the OmniCopilot panel) uses to show a key holder their spend. + +```bash +# Text form (the historical contract — plain text for a terminal) +curl -H "Authorization: Bearer " \ + http://localhost:20128/api/usage/om-usage + +# Structured form — what a UI consumes +curl -H "Authorization: Bearer " \ + "http://localhost:20128/api/usage/om-usage?format=json" +``` + +The key must have **`allowUsageCommand`** enabled (off by default — the dashboard's API-key +manager toggles it per key). Without it the endpoint answers `403`. + +`?format=json` returns a discriminated shape so a caller never reads a data field off a +refusal. On success: + +```jsonc +{ + "allowed": true, + // present only when the key opted into per-key usage limits (daily/weekly USD): + "personal": { "dailySpentUsd": 1.25, "dailyLimitUsd": 5, "dailyResetAtIso": "…", "weeklySpentUsd": 8, "weeklyLimitUsd": 20, "weeklyResetAtIso": "…" /* … */ }, + // the selected provider quota snapshot, or null when nothing is cached yet: + "provider": { "connectionId": "…", "provider": "claude", "plan": "…", "quotas": { /* … */ } }, + // every connection's snapshot, so a UI can render several providers side by side: + "providers": [ { "connectionId": "…", "provider": "claude", /* … */ }, { "provider": "codex", /* … */ } ] +} +``` + +On refusal (`401` bad key / `403` not allowed) the same route returns +`{ "allowed": false, "error": { "message": "…" } }` — a present-but-empty `personal`/`provider` +(key allowed, nothing learned yet) is a different state from a refusal, and only the JSON form +distinguishes them. + +**Auth:** the caller's own Bearer API key, validated with `isValidApiKey` — this is *not* the +management surface (`/api/keys/…`), which stays behind `requireManagementAuth`. + +--- + ## Semantic Cache ```bash diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index ae06e620905..068c2b02066 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -772,6 +772,7 @@ REQUEST_TIMEOUT_MS (global override) | `KIMI_WEB_BASE_URL` | `https://www.kimi.ai` | Base URL for the Kimi Web (international kimi.ai Connect-RPC) executor (`kimi-web.ts`); override only for mirror/proxy endpoints. | | `KIMI_WEB_CHAT_URL` | `/apiv2/kimi.gateway.chat.v1.ChatService/Chat` | Full chat endpoint for the Kimi Web executor (`kimi-web.ts`). | | `OMNIROUTE_LOGIN_BROWSER_PATH` | _(auto-detected)_ | Path to a system Chrome/Edge executable for the Adobe Firefly interactive browser sign-in (`adobeFireflyBrowserLogin.ts`); overrides per-OS auto-detection. | +| `OMNIROUTE_STANDALONE_DIR` | _.build/ standalone output_ | Build-time override for the standalone output directory consumed by the post-build colocation step (`scripts/build/colocate-standalone.mjs`); build tooling, not runtime. | Combo target attempts inherit the resolved upstream request timeout (`FETCH_TIMEOUT_MS`, or `REQUEST_TIMEOUT_MS` when it supplies the fetch default). Set `targetTimeoutMs` in a combo, diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index 8a0a3b5d99a..571fe0e904c 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -1,16 +1,16 @@ --- title: "Provider Reference" version: 3.8.50 -lastUpdated: 2026-08-22 +lastUpdated: 2026-08-21 --- # Provider Reference > **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. > Regenerate with: `npm run gen:provider-reference` -> **Last generated:** 2026-08-22 +> **Last generated:** 2026-08-21 -Total providers: **348**. See category breakdown below. +Total providers: **349**. See category breakdown below. ## Categories @@ -34,7 +34,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each --- -## No-auth Providers (no key required) (12) +## No-auth Providers (no key required) (11) | ID | Alias | Name | Tags | Website | Notes | Tool calling | |----|-------|------|------|---------|-------|--------------| @@ -47,7 +47,6 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — | | `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — | | `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — | -| `uncloseai` | `unc` | UncloseAI | No-auth | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | — | | `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — | | `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — | @@ -99,7 +98,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — | | `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated | | `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — | -| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://chat.minimax.io) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | +| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | | `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — | | `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — | | `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated | @@ -121,7 +120,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | | `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | -## API Key Providers (paid / paid-with-free-credits) (231) +## API Key Providers (paid / paid-with-free-credits) (233) | ID | Alias | Name | Tags | Website | Notes | |----|-------|------|------|---------|-------| @@ -150,7 +149,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | | `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | | `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | -| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | ⚠️ **DEPRECATED.** api.blackbox.ai returns HTTP 404 on every path variant (sweep 2026-08-21); the public inference surface has moved to the gated enterprise.blackbox.ai/v1 endpoint. | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply | | `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | | `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | | `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | @@ -167,7 +166,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — | | `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | | `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | -| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /provider/v1/chat/completions endpoint. | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | | `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | | `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | | `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. | @@ -214,7 +213,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | | `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | | `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | -| `hackclub` | `hc` | Hackclub AI | API key | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | | `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | | `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | | `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | @@ -243,6 +242,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. | | `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | | `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | +| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. | | `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | | `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | | `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | @@ -332,6 +332,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | | `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. | | `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | | `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. | | `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | | `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | diff --git a/llm.txt b/llm.txt index 51feb87bc32..ac68d5cb4e4 100644 --- a/llm.txt +++ b/llm.txt @@ -1,6 +1,6 @@ # OmniRoute -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -165,7 +165,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -277,7 +277,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -475,7 +475,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/open-sse/config/constants.ts b/open-sse/config/constants.ts index f1fa055c938..d99da4725b0 100644 --- a/open-sse/config/constants.ts +++ b/open-sse/config/constants.ts @@ -355,6 +355,23 @@ export const STREAM_RECOVERY = { HOLDBACK_MS: 750, BUFFER_MAX_BYTES: 65536, EARLY_RETRY_MAX: 4, + /** + * Minimum character overlap `trimContinuationOverlap` must find between the + * already-emitted text and a mid-stream continuation for the continuation to be + * accepted as a real resume, rather than an unrelated restart the model produced after + * ignoring the assistant-prefill. + * + * This is a DOCUMENTED TRADE-OFF, not a solved distinction: a model that continues + * cleanly with fewer than this many echoed characters (a legitimate, even preferred, + * outcome — there was nothing to de-duplicate) is indistinguishable, from string data + * alone, from a model that silently restarted on an unrelated sentence. Both produce a + * low/zero overlap. Rejecting below this threshold trades some false-positive rejections + * of legitimate low-overlap continuations (bounded retry, then a clean close — no data + * loss beyond that retry) against not silently gluing two unrelated fragments into one + * corrupted, unrecoverable answer. It does not eliminate the residual false negative + * either (an accidental coincidence at or above this many characters is still accepted). + */ + MIN_CONTINUATION_OVERLAP_CHARS: 8, } as const; /** diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 9c1580e7aed..8668de2c636 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -19,17 +19,16 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ export const GLM_SHARED_MODELS = Object.freeze([ { - // GLM-5.3 (2026-08-14): one upstream id; effort is the reasoning_effort - // param (low|high|max, default max) — the -high/-low entries below are - // OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. - // Default context window not yet published by Z.ai; 1M mirrored from - // GLM-5.2 (same base model). https://z.ai/blog/glm-5.3 + // GLM-5.3 exposes low|high|max reasoning_effort (default max); -high/-low + // are OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. + // https://docs.z.ai/guides/llm/glm-5.3 id: "glm-5.3", name: "GLM 5.3", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], }, { id: "glm-5.3-high", @@ -38,6 +37,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.3-low", @@ -46,14 +46,19 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low"], }, { + // GLM-5.2 has two positive effective tiers: low/medium map to high and xhigh + // maps to max; disabling thinking remains the separate thinking toggle. + // https://docs.z.ai/guides/capabilities/thinking id: "glm-5.2", name: "GLM 5.2", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high", "max"], }, { id: "glm-5.2-high", @@ -62,6 +67,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.2-max", @@ -70,14 +76,18 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["max"], }, { + // Earlier GLM families support the thinking toggle, not reasoning_effort. + // An explicit empty list prevents generic catalog tiers from being inferred. id: "glm-5.1", name: "GLM 5.1", contextLength: 204800, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5", @@ -86,6 +96,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5-turbo", @@ -94,6 +105,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7-flash", @@ -102,6 +114,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7", @@ -110,6 +123,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.6v", @@ -118,6 +132,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -127,6 +142,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5v", @@ -135,6 +151,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -144,6 +161,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5-air", @@ -152,6 +170,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, ]); diff --git a/open-sse/config/providers/index.ts b/open-sse/config/providers/index.ts index b0135fa5242..3e2c812e851 100644 --- a/open-sse/config/providers/index.ts +++ b/open-sse/config/providers/index.ts @@ -264,6 +264,7 @@ import { freeAiProvider } from "./registry/free-ai/index.ts"; import { voidAiProvider } from "./registry/void-ai/index.ts"; import { helixmindProvider } from "./registry/helixmind/index.ts"; import { tabitokenProvider } from "./registry/tabitoken/index.ts"; +import { logfareProvider } from "./registry/logfare/index.ts"; export const REGISTRY: Record = { aimlapi: aimlapiProvider, @@ -532,4 +533,5 @@ export const REGISTRY: Record = { "void-ai": voidAiProvider, helixmind: helixmindProvider, tabitoken: tabitokenProvider, + logfare: logfareProvider, }; diff --git a/open-sse/config/providers/registry/logfare/index.ts b/open-sse/config/providers/registry/logfare/index.ts new file mode 100644 index 00000000000..9b16b5a2f4e --- /dev/null +++ b/open-sse/config/providers/registry/logfare/index.ts @@ -0,0 +1,25 @@ +import type { RegistryEntry } from "../../shared.ts"; +import { buildOpenAiCompatibleRegistryEntry } from "../../shared.ts"; + +/** + * Logfare — free OpenAI-compatible LLM inference provider. + * + * Live-verified 2026-08-21: GET https://logfare.ai/v1/models returns a real + * catalog (20 models; 11 chat-capable incl. kimi-k3, deepseek-v4-pro, + * glm-5.2, gpt-5.6-luna, minimax-m3). Auth is a Bearer API key issued + * instantly at https://logfare.ai/register (username/password, no email). + * + * ⚠️ Privacy: in exchange for free inference, Logfare logs every request + * (prompts, completions, metadata). After PII scrubbing this may feed their + * private internal evaluation datasets. Users can opt out at /consent; see + * https://logfare.ai/tos and https://logfare.ai/privacy. The dashboard card + * surfaces this via freeNote. + */ +export const logfareProvider: RegistryEntry = buildOpenAiCompatibleRegistryEntry({ + id: "logfare", + alias: "logfare", + baseUrl: "https://logfare.ai/v1/chat/completions", + modelsUrl: "https://logfare.ai/v1/models", + models: [], + passthroughModels: true, +}); diff --git a/open-sse/config/providers/registry/uncloseai/index.ts b/open-sse/config/providers/registry/uncloseai/index.ts index baea064e3c6..2a7b59f5865 100644 --- a/open-sse/config/providers/registry/uncloseai/index.ts +++ b/open-sse/config/providers/registry/uncloseai/index.ts @@ -6,6 +6,7 @@ export const uncloseaiProvider: RegistryEntry = { format: "openai", executor: "default", baseUrl: "https://hermes.ai.unturf.com/v1/chat/completions", + modelsUrl: "https://hermes.ai.unturf.com/v1/models", authType: "optional", authHeader: "bearer", models: [ diff --git a/open-sse/config/providers/registry/zcode/index.ts b/open-sse/config/providers/registry/zcode/index.ts index cd2a4eece64..65e2e1c3d3d 100644 --- a/open-sse/config/providers/registry/zcode/index.ts +++ b/open-sse/config/providers/registry/zcode/index.ts @@ -1,6 +1,17 @@ import type { RegistryEntry } from "../../shared.ts"; import { GLM_SHARED_MODELS } from "../../../glmProvider.ts"; +const GLM_EXECUTOR_EFFORT_ALIASES = new Set([ + "glm-5.3-high", + "glm-5.3-low", + "glm-5.2-high", + "glm-5.2-max", +]); + +export const ZCODE_MODELS = GLM_SHARED_MODELS.filter( + (model) => !GLM_EXECUTOR_EFFORT_ALIASES.has(model.id) +).map((model) => ({ ...model, supportedThinkingEfforts: [] })); + /** * Local ZCode app-server backend. Authentication remains in the user's local * ZCode profile (`builtin:zai-coding-plan`); OmniRoute does not receive or @@ -14,5 +25,7 @@ export const zcodeProvider: RegistryEntry = { baseUrl: "zcode://app-server/stdio", authType: "none", authHeader: "none", - models: [...GLM_SHARED_MODELS], + // ZCode's app-server transport does not consume reasoning_effort; keep thinking + // capability metadata without advertising aliases or tiers that it would ignore. + models: ZCODE_MODELS, }; diff --git a/open-sse/executors/accountRotation.ts b/open-sse/executors/accountRotation.ts index a64321e83d2..67115bdd8d0 100644 --- a/open-sse/executors/accountRotation.ts +++ b/open-sse/executors/accountRotation.ts @@ -1,7 +1,7 @@ /** * Shared multi-account rotation mechanics for noauth executors that round-robin * across several "accounts" (fingerprints), each with an optional dedicated - * proxy — currently `OpencodeExecutor` and `MimocodeExecutor`. + * proxy — currently `OpencodeExecutor`. * * Extracted after both executors independently implemented the same * pickAccount/markCooldown/markSuccess skeleton with the same exponential @@ -120,3 +120,58 @@ export function maskAccountId(fingerprint: string): string { export function isNetworkErrorRotatable(account: RotatableAccount): boolean { return account.proxy !== null; } + +/** + * Detect an *empty* upstream rejection: a 400 whose body carries no usable + * completion — the kind `OpencodeExecutor` must rotate/retry on instead of + * propagating as a fatal success. + * + * Signature is deliberately strict and scoped to the observed malformed + * envelope (`choices[0].message` with no `error`, no real `content`, + * `finish_reason: null`): + * - status must be exactly 400 (anything else → false); + * - body must parse and contain a `choices` array with at least one entry + * holding a `message` object; + * - an `error` field (present or empty) → false, so genuine 400s keep + * propagating immediately (#10460 precedent: classify by signature before + * rotating); + * - `tool_calls` / `reasoning_content` → false (real content); + * - `message.content` absent / null / "" → eligible; any other value + * (non-empty text, number, block array…) → false (conservative); + * - a literal `finish_reason` (not null) → false (a completed, if empty, turn). + * + * Does NOT reuse `detectMalformedNonStream` (diagnostics.ts): that classifier + * also flags `{error:{…}}` bodies as `empty_choices`, which would rotate on + * real errors — a false-positive class with a history here. + */ +export function isEmptyUpstreamRejection(status: number, bodyText: string): boolean { + if (status !== 400) return false; + let parsed: unknown; + try { + parsed = JSON.parse(bodyText); + } catch { + return false; + } + const choices = (parsed as { choices?: unknown })?.choices; + if (!Array.isArray(choices) || choices.length === 0) return false; + const first = choices[0] as { message?: unknown; finish_reason?: unknown }; + if (typeof first !== "object" || first === null) return false; + const rawMessage = (first as { message?: unknown }).message; + if (typeof rawMessage === "undefined" || rawMessage === null) return false; + if (typeof parsed !== "object" || parsed === null) return false; + if ("error" in (parsed as Record)) return false; + const msg = rawMessage as Record; + if ("tool_calls" in msg) return false; + if ("reasoning_content" in msg) return false; + const content = msg.content; + if (content !== undefined && content !== null && content !== "") return false; + if (first.finish_reason !== null && first.finish_reason !== undefined) return false; + return true; +} + +/** Best-effort extraction of the upstream `chatcmpl_*` id from a response body, + * for observability logging. Returns `"unknown"` when absent or unparseable. */ +export function extractChatcmplId(bodyText: string): string { + const match = /"id"\s*:\s*"(chatcmpl_[^"]+)"/.exec(bodyText); + return match ? match[1] : "unknown"; +} diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index e1674c30fb3..b9f6cf113a2 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -517,10 +517,12 @@ export function filterNonstandardCodexSse(response: Response): Response { const transform = new TransformStream({ transform(chunk, controller) { buffer += decoder.decode(chunk, { stream: true }); - let sep: number; - while ((sep = buffer.indexOf("\n\n")) !== -1) { - const block = buffer.slice(0, sep + 2); - buffer = buffer.slice(sep + 2); + while (true) { + const separator = /\r?\n\r?\n/.exec(buffer); + if (!separator) break; + const blockEnd = separator.index + separator[0].length; + const block = buffer.slice(0, blockEnd); + buffer = buffer.slice(blockEnd); if (!dropBlock(block)) controller.enqueue(encoder.encode(block)); } }, @@ -1396,7 +1398,6 @@ export class CodexExecutor extends BaseExecutor { provider: "codex", preserveEncryptedReasoning: credentials?.providerSpecificData?.preserveEncryptedReasoning === true, - onIncompatibleReasoning: "drop", }); if (nativeCodexPassthrough) { diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index a222571022a..eda82962557 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -73,7 +73,7 @@ type GlmEffortTier = { * `thinking.type=enabled` (5.3 no longer accepts thinking disabled). * * https://docs.z.ai/devpack/latest-model - * https://z.ai/blog/glm-5.3 + * https://docs.z.ai/guides/llm/glm-5.3 */ function parseGlmEffortTier(model: string): GlmEffortTier | null { switch (model) { @@ -399,7 +399,24 @@ export class GlmExecutor extends DefaultExecutor { ): Promise { const credentials = input.credentials; const url = buildGlmChatUrl(credentials?.providerSpecificData, transport, this.config.baseUrl); - const headers = this.buildHeaders(credentials, input.stream, input.clientHeaders, input.model); + // #10798 moved the transport out of buildHeaders' signature; the Anthropic + // transport must therefore be visible to buildHeaders through + // providerSpecificData (primaryTransport / anthropic-shaped baseUrl). + const headers = + transport === "anthropic" + ? this.buildHeaders( + { + ...credentials, + providerSpecificData: { + ...credentials?.providerSpecificData, + primaryTransport: "anthropic", + }, + }, + input.stream, + input.clientHeaders, + input.model + ) + : this.buildHeaders(credentials, input.stream, input.clientHeaders, input.model); applyConfiguredUserAgent(headers, credentials.providerSpecificData); mergeUpstreamExtraHeaders(headers, input.upstreamExtraHeaders); diff --git a/open-sse/executors/opencode.ts b/open-sse/executors/opencode.ts index c00ae258a3d..0829bfa871d 100644 --- a/open-sse/executors/opencode.ts +++ b/open-sse/executors/opencode.ts @@ -15,6 +15,8 @@ import { markSuccess as markAccountSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, } from "./accountRotation.ts"; import { isNetworkRotationSharedEgressGuardEnabled } from "@/shared/utils/featureFlags"; @@ -253,14 +255,41 @@ export class OpencodeExecutor extends BaseExecutor { try { this.syncAccountsFromCredentials(input.credentials); + const { log } = input; const hasProxies = this.accounts.some((a) => a.proxy !== null); - // Fast path: no multi-account proxy wiring configured → original behavior. + // Fast path: no multi-account proxy wiring configured → original behavior, + // plus exactly ONE bounded retry when the upstream answers a 400 empty + // rejection (same predicate and logging as the rotation loop). Everything + // else passes untouched: this path deliberately preserves BaseExecutor's + // intra-URL 429 retries (no skipUpstreamRetry here). if (this.accounts.length === 1 && !hasProxies) { - return await super.execute(input); + const single = (await super.execute(input)) as HttpExecuteResult; + if (single.response.status === 400) { + let bodyText: string | null = null; + try { + bodyText = await single.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on direct account"); + } + if (bodyText !== null) { + if (isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on direct account (${chatcmplId}), retrying once…` + ); + return await super.execute(input); + } + log?.debug?.( + "OPENCODE", + "400 without error field, signature not matched on direct account — observing" + ); + } + } + return single; } - const { log } = input; // This loop only ever dispatches through super.execute() (the HTTP request // path), which always resolves the object-shaped arm of ExecutorExecuteResult // — the bare-Response arm belongs to web/scraping executors only (base.ts:290). @@ -277,8 +306,13 @@ export class OpencodeExecutor extends BaseExecutor { // network call, but proxied accounts (independent egress) are still // tried normally. let sharedEgressDown = false; + // Bounded extra attempts for empty upstream rejections: +1 for a single + // account (retry the same one), none for a multi-account fleet (rotation + // through the accounts is the retry). Avoids an unbounded loop on a + // persistently malformed upstream. + const emptyRejectionBudget = this.accounts.length === 1 ? 1 : 0; - for (let attempt = 0; attempt < this.accounts.length; attempt++) { + for (let attempt = 0; attempt < this.accounts.length + emptyRejectionBudget; attempt++) { const account = this.pickAccount(); const masked = maskAccountId(account.fingerprint); @@ -354,6 +388,34 @@ export class OpencodeExecutor extends BaseExecutor { continue; } + // Empty upstream rejection (malformed 400: no error field, no real + // content, finish_reason null — see isEmptyUpstreamRejection). Rotate/ + // retry instead of propagating it as a fatal success: the observed + // envelope was marking subagent sessions as failed. Read the body ONLY + // for a 400 (never a 200/streaming — that would buffer the good path); + // classify, log, and continue. Neitheries markCooldown nor markSuccess: + // the failure is upstream's, not this account's. + if (status === 400) { + let bodyText: string | null = null; + try { + bodyText = await result.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on empty rejection check"); + } + if (bodyText !== null && isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on account ${masked} (${chatcmplId}), rotating to next…` + ); + continue; + } + // A 400 carrying a real error (or non-empty content): propagate + // immediately, untouched — same as before this change. + this.markSuccess(account); + return result; + } + this.markSuccess(account); return result; } diff --git a/open-sse/executors/zcode.ts b/open-sse/executors/zcode.ts index 0841b4daa8e..8f0a98ac142 100644 --- a/open-sse/executors/zcode.ts +++ b/open-sse/executors/zcode.ts @@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto"; import { existsSync } from "node:fs"; import { homedir } from "node:os"; import { join, resolve } from "node:path"; -import { GLM_SHARED_MODELS } from "../config/glmProvider.ts"; +import { ZCODE_MODELS } from "../config/providers/registry/zcode/index.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorExecuteResult, type ProviderCredentials } from "./base.ts"; import { ZcodeAppServerClient, type ZcodeClientLike } from "./zcodeProtocol.ts"; import { buildErrorBody, errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; @@ -12,8 +12,8 @@ const DEFAULT_PROVIDER_ID = "builtin:zai-coding-plan"; const DEFAULT_TURN_TIMEOUT_MS = 120_000; const DEFAULT_POLL_INTERVAL_MS = 250; const TERMINAL_STATUSES = new Set(["completed", "idle", "paused", "error"]); -const ZCODE_MODEL_ALLOWLIST = new Set(GLM_SHARED_MODELS.map((model) => model.id)); -const DEFAULT_ZCODE_MODEL = GLM_SHARED_MODELS[0]?.id || "glm-5.2"; +const ZCODE_MODEL_ALLOWLIST = new Set(ZCODE_MODELS.map((model) => model.id)); +const DEFAULT_ZCODE_MODEL = ZCODE_MODELS[0]?.id || "glm-5.2"; type JsonRecord = Record; type OpenAIMsg = { role?: string; content?: unknown }; diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 8f11373b1fb..fea8640f9d3 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -516,7 +516,7 @@ export async function handleChatCore({ conversationId = null, modelPinned = false, skipResourcePressureGuard = false, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", managedLease = null, }) { let { provider, model, extendedContext } = modelInfo; @@ -1213,7 +1213,7 @@ export async function handleChatCore({ provider, preserveEncryptedReasoning: credentials?.providerSpecificData?.preserveEncryptedReasoning === true, - onIncompatibleReasoning: reasoningTransportFallback === "drop" ? "drop" : "reject", + onIncompatibleReasoning: reasoningTransportFallback === "skip" ? "reject" : "drop", } ); if (policy.incompatibleReasoning) { diff --git a/open-sse/handlers/imageGeneration/providers/aihorde.ts b/open-sse/handlers/imageGeneration/providers/aihorde.ts index 4d39270c034..13fa50f5abf 100644 --- a/open-sse/handlers/imageGeneration/providers/aihorde.ts +++ b/open-sse/handlers/imageGeneration/providers/aihorde.ts @@ -103,11 +103,12 @@ async function fetchHordeImageBytes( if (value.startsWith("http://") || value.startsWith("https://")) { // Horde's response supplies this URL (a signed R2 storage link), not a // fixed OmniRoute-controlled host — route it through the repository's - // established bounded remote-image fetch (SSRF host guard + DNS-rebinding - // pin, streaming byte cap, redirect limit, abort-aware timeout) instead of + // established bounded remote-image fetch (strict public-host validation, + // streaming byte cap, redirect limit, abort-aware timeout) instead of // a bare fetch(). Same helper `imageGeneration.ts` already uses for other // providers' remote image URLs. const remote = await fetchRemoteImage(value, { + guard: "public-only", timeoutMs: options.timeoutMs, signal: options.signal ?? undefined, maxBytes: MAX_HORDE_IMAGE_BYTES, diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index 8b7f3cee324..50c16fa9665 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -106,6 +106,29 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("declares exact GLM reasoning-effort tiers across every shared GLM provider", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt"]) { + for (const model of getModelsByProviderId(provider)) { + expect(model.supportedThinkingEfforts, `${provider}/${model.id} effort tiers`).toEqual( + routedTiers.get(model.id) ?? [] + ); + } + } + + for (const model of getModelsByProviderId("zcode")) { + expect(model.supportedThinkingEfforts, `zcode/${model.id} effort tiers`).toEqual([]); + } + }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); diff --git a/open-sse/services/antigravityIdentity.ts b/open-sse/services/antigravityIdentity.ts index f3934e84605..3703cd9c35b 100644 --- a/open-sse/services/antigravityIdentity.ts +++ b/open-sse/services/antigravityIdentity.ts @@ -75,7 +75,6 @@ export function getAntigravitySessionId( fallback?: unknown ): string { return ( - deriveAntigravitySessionId(getAntigravityAccountKey(credentials)) || toNonEmptyString(fallback) || generateAntigravitySessionId() ); diff --git a/open-sse/services/autoCombo/virtualFactory.ts b/open-sse/services/autoCombo/virtualFactory.ts index 28071144fbb..e3dbac4d73c 100644 --- a/open-sse/services/autoCombo/virtualFactory.ts +++ b/open-sse/services/autoCombo/virtualFactory.ts @@ -7,6 +7,8 @@ import { getProviderRegistry } from "./providerRegistryAccessor"; import type { ConnectionFields } from "@/lib/db/encryption"; import { NOAUTH_PROVIDERS } from "@/shared/constants/providers"; import { hasUsableWebSessionCredential } from "@/shared/providers/webSessionCredentials"; +import { toNumber } from "@/shared/utils/numeric"; +import { isCompatibleProviderConnectionId } from "@/shared/utils/compatibleProviderId"; import { defaultLogger as log } from "@omniroute/open-sse/utils/logger"; import { getTokenLimit } from "../contextManager"; import { @@ -179,9 +181,31 @@ function hasProviderSpecificSessionData(conn: VirtualFactoryConn): boolean { return hasUsableWebSessionCredential(conn.provider, conn.providerSpecificData); } +/** + * #11180: a custom compatible connection (`openai-compatible-*` / + * `anthropic-compatible-*`) may legitimately carry no credential at all, + * because it points at a self-hosted backend the operator started without one + * (`llama-server --host 0.0.0.0` with no `--api-key`, Ollama, vLLM). For those + * IDs "no credential" is the normal configuration rather than an unconfigured + * connection, so the credential gate must not silently drop them from every + * `auto/*` pool while direct `/` calls keep working. + * + * Deliberately narrow: only the four generated compatible-provider ID shapes + * qualify. A first-party provider with an empty key really is unconfigured and + * stays filtered out, and the no-auth registry allowlist below is untouched. + */ +function isKeylessEligibleConnection(conn: VirtualFactoryConn): boolean { + return isCompatibleProviderConnectionId(conn.provider); +} + function hasUsableConnectionCredential(conn: VirtualFactoryConn): boolean { const hasApiKey = typeof conn.apiKey === "string" && conn.apiKey.trim().length > 0; - return hasApiKey || hasUsableOAuthToken(conn) || hasProviderSpecificSessionData(conn); + return ( + hasApiKey || + hasUsableOAuthToken(conn) || + hasProviderSpecificSessionData(conn) || + isKeylessEligibleConnection(conn) + ); } const SYNTHETIC_NOAUTH_CONNECTION_ID = RESILIENCE_NOAUTH_CONNECTION_ID; @@ -607,7 +631,7 @@ export async function prepareVirtualAutoComboInputs( // remaining allowance as a percentage, and a raw ">0" comparison would // let a reading of e.g. 0.3% (rounding noise, not real headroom) pass. minRemainingAllowance: 1, - maxStateAgeMs: (Number(settings.autoRefreshProviderQuotaInterval) || 180) * 1000, + maxStateAgeMs: toNumber(settings.autoRefreshProviderQuotaInterval, 180) * 1000, }); if (strictFilteredPool !== pool) pool = strictFilteredPool; diff --git a/open-sse/services/cloudCodeThinking.ts b/open-sse/services/cloudCodeThinking.ts index 443bc6510ec..b9c3e434a15 100644 --- a/open-sse/services/cloudCodeThinking.ts +++ b/open-sse/services/cloudCodeThinking.ts @@ -6,11 +6,10 @@ function isRecord(value: unknown): value is Record { return !!value && typeof value === "object" && !Array.isArray(value); } +const PREFIX_TRIM_RE = /^(?:models\/|antigravity\/)+/i; + function normalizeCloudCodeModel(model: string): string { - return String(model || "") - .trim() - .replace(/^models\//i, "") - .replace(/^antigravity\//i, ""); + return String(model || "").trim().replace(PREFIX_TRIM_RE, ""); } function stripGeminiThinkingConfig(value: unknown): unknown { diff --git a/open-sse/services/combo/resolveAutoStrategy.ts b/open-sse/services/combo/resolveAutoStrategy.ts index 7d6fe2a682d..45ef4b5c605 100644 --- a/open-sse/services/combo/resolveAutoStrategy.ts +++ b/open-sse/services/combo/resolveAutoStrategy.ts @@ -311,6 +311,12 @@ export async function resolveAutoStrategyOrder( taskType, requestHasTools, lastKnownGoodProvider, + // #11181: the Routing tab persists an LKGP on/off toggle and + // LKGPStrategy guards on `context.lkgpEnabled === false`, but the + // field was never forwarded into this context, so the guard never + // saw the setting and the off-switch was unreachable. + lkgpEnabled: (settings as { lkgpEnabled?: unknown } | null | undefined)?.lkgpEnabled as + boolean | undefined, estimatedInputTokens, sla: slaPolicy, }, diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index 7c7f0ce8a92..76ef5546aba 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -617,7 +617,18 @@ export async function validateResponseQuality( try { json = JSON.parse(text); } catch { - if (text.startsWith("data:") || text.startsWith("event:")) return { valid: true }; + // An SSE stream body is expected for streamed upstreams. Besides `data:` and + // `event:` frames, the SSE spec also allows comment lines that begin with a + // colon (`:`), which providers use for keep-alives while the model is still + // generating — e.g. OpenRouter emits `: OPENROUTER PROCESSING` on slower / + // reasoning responses. A stream that opens with such a comment (or with + // leading whitespace/newlines) is still a valid stream, not malformed JSON, + // so trim and recognize the comment prefix before rejecting. Without this, + // otherwise-good streamed completions get failed as "not valid JSON". + const trimmed = text.trimStart(); + if (trimmed.startsWith("data:") || trimmed.startsWith("event:") || trimmed.startsWith(":")) { + return { valid: true }; + } return { valid: false, reason: "response is not valid JSON" }; } diff --git a/open-sse/services/compression/engines/ccr/index.ts b/open-sse/services/compression/engines/ccr/index.ts index d2136fc8f19..92869bfbc9b 100644 --- a/open-sse/services/compression/engines/ccr/index.ts +++ b/open-sse/services/compression/engines/ccr/index.ts @@ -45,7 +45,7 @@ import { } from "../../../../../src/lib/db/ccrBlocks.ts"; import { createCompressionStats } from "../../stats.ts"; import { queryBlock, type CcrQuery } from "./ccrQuery.ts"; -import { injectCcrProtocolInstruction } from "./protocolInstruction.ts"; +import { callerSupportsCcrRetrieve, injectCcrProtocolInstruction } from "./protocolInstruction.ts"; import type { CompressionEngine, CompressionEngineApplyOptions, @@ -939,6 +939,30 @@ export const ccrEngine: CompressionEngine = { return { body, compressed: false, stats: null }; } + // #7746 follow-up: only callers whose tools[] proves they can reach + // omniroute_ccr_retrieve may have content replaced at all. For everyone + // else (plain OpenAI-compatible clients — the marker is an MCP-only + // contract) replacement would strand the original text behind a hash the + // model has no way to resolve. Skip the whole engine for them. The check + // is wrapped defensively: a malformed body must fail OPEN (no + // compression), never throw into the request pipeline. + let callerCanRetrieve = false; + try { + callerCanRetrieve = callerSupportsCcrRetrieve(body); + } catch (err) { + // Defensive: the helper is total, but if it ever throws we must fail + // OPEN (no compression) — and surface it so a future regression in the + // helper is visible instead of silently bypassing compression forever. + console.warn( + "[compression/ccr] callerSupportsCcrRetrieve threw; skipping compression:", + err instanceof Error ? err.message : err + ); + callerCanRetrieve = false; + } + if (!callerCanRetrieve) { + return { body, compressed: false, stats: null }; + } + const minChars = typeof stepConfig["minChars"] === "number" ? (stepConfig["minChars"] as number) diff --git a/open-sse/services/contextManager.ts b/open-sse/services/contextManager.ts index a2d678f107c..6fe9e94c8e2 100644 --- a/open-sse/services/contextManager.ts +++ b/open-sse/services/contextManager.ts @@ -669,13 +669,35 @@ function purifyHistory(messages: Record[], targetTokens: number result = fixToolPairs(result); result = stripTrailingAssistantOrphanToolUse(result); - // Add summary of dropped messages + // Add summary of dropped messages. Merge the notice INTO the leading + // system/developer message instead of splicing a second system-role message + // mid-array: strict gateways (TokenRouter confirmed live 2026-08-22, see the + // PROVIDERS_SYSTEM_MUST_BE_FIRST list in src/lib/memory/injection.ts) reject + // any system message at index > 0 with HTTP 400 "System message must be at + // the beginning". When there is no leading system message, prepend one -- + // index 0 is accepted by every provider (same slot the old splice used when + // system[] was empty). if (keep < nonSystem.length) { const dropped = nonSystem.length - keep; - result.splice(system.length, 0, { - role: "system", - content: `[Context compressed: ${dropped} earlier messages removed to fit context window]`, - }); + const droppedNotice = `[Context compressed: ${dropped} earlier messages removed to fit context window]`; + const first = result[0]; + if (first && (first.role === "system" || first.role === "developer")) { + if (typeof first.content === "string") { + result[0] = { + ...first, + content: first.content ? `${droppedNotice}\n${first.content}` : droppedNotice, + }; + } else if (Array.isArray(first.content)) { + result[0] = { + ...first, + content: [{ type: "text", text: droppedNotice }, ...(first.content as unknown[])], + }; + } else { + result[0] = { ...first, content: droppedNotice }; + } + } else { + result.unshift({ role: "system", content: droppedNotice }); + } } return result; diff --git a/open-sse/services/rateLimitManager.ts b/open-sse/services/rateLimitManager.ts index a7bdeb2b09b..18815b32a5e 100644 --- a/open-sse/services/rateLimitManager.ts +++ b/open-sse/services/rateLimitManager.ts @@ -756,7 +756,7 @@ export function updateFromHeaders(provider, connectionId, headers, status, model ); } else if (remaining > limit * 0.5) { // Plenty of headroom — relax the limiter - updates.minTime = 0; + updates.minTime = resolveMinTime(currentRequestQueueSettings.minTimeBetweenRequestsMs); updates.reservoir = null; updates.reservoirRefreshAmount = null; updates.reservoirRefreshInterval = null; diff --git a/open-sse/services/reasoningInputPolicy.ts b/open-sse/services/reasoningInputPolicy.ts index 94f27ddcdc9..0e9e8d194db 100644 --- a/open-sse/services/reasoningInputPolicy.ts +++ b/open-sse/services/reasoningInputPolicy.ts @@ -37,7 +37,6 @@ export interface ReasoningInputPolicyOptions { export interface ReasoningInputPolicyResult { incompatibleReasoning: boolean; } - export function resolveReasoningTransport( provider: string | null | undefined, preserveEncryptedReasoning = false @@ -47,18 +46,6 @@ export function resolveReasoningTransport( return transport ?? (preserveEncryptedReasoning ? "opaque" : "plaintext"); } -export function createReasoningTransportIncompatibleError(): Error & { - statusCode: number; - errorType: string; -} { - const error = new Error( - "Reasoning continuation is not compatible with the selected target" - ) as Error & { statusCode: number; errorType: string }; - error.statusCode = 400; - error.errorType = "reasoning_transport_incompatible"; - return error; -} - function asRecord(value: unknown): JsonRecord | null { return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null; } @@ -101,11 +88,12 @@ function hasChatPlaintextReasoning(record: JsonRecord): boolean { /** * Returns only provider-authentic plaintext continuation state. Display summaries - * are excluded, and a record carrying opaque state is never cross-converted. + * and opaque-only records are excluded. Explicit plaintext remains independently + * portable when the same record also carries an opaque companion (#10949). */ export function extractReplayableResponsesReasoningText(value: unknown): string { const record = asRecord(value); - if (!record || record.type !== "reasoning" || hasOpaqueReasoningState(record)) return ""; + if (!record || record.type !== "reasoning") return ""; if (!Array.isArray(record.content)) return ""; return record.content @@ -307,9 +295,9 @@ function sanitizeResponsesInput( } /** - * Applies one protocol-independent compatibility decision before request translation. - * Plaintext is portable by default; opaque state requires an explicit target declaration. - * Display summaries do not affect compatibility; stateless input drops orphan summaries. + * Projects reasoning continuation onto the selected target transport. + * Incompatible active state is dropped by default; combo routing may reject an + * attempt instead so it can fall through without mutating the request. */ export function applyReasoningInputPolicy( body: Record, @@ -321,14 +309,17 @@ export function applyReasoningInputPolicy( inputFormat === "responses" ? inspectResponsesReasoning(body.input) : inspectChatReasoning(body.messages); - const incompatibleReasoning = !isReasoningCompatible(inspection, transport); + const mixedState = inspection.hasPlaintext && inspection.hasOpaque; + const incompatibleReasoning = !mixedState && !isReasoningCompatible(inspection, transport); + // Mixed plaintext + opaque input (#10949) is never a rejection: it is projected + // onto the target transport by the per-item sanitizers below. - if (incompatibleReasoning && options.onIncompatibleReasoning !== "drop") { + if (incompatibleReasoning && options.onIncompatibleReasoning === "reject") { return { incompatibleReasoning: true }; } if (inputFormat === "chat") { - if (incompatibleReasoning && Array.isArray(body.messages)) { + if ((incompatibleReasoning || mixedState) && Array.isArray(body.messages)) { body.messages = dropIncompatibleChatReasoning(body.messages, transport); } return { incompatibleReasoning: false }; @@ -343,12 +334,13 @@ export function applyReasoningInputPolicy( }, ]; } - if (!Array.isArray(body.input)) return { incompatibleReasoning: false }; - body.input = sanitizeResponsesInput( - body.input, - transport, - incompatibleReasoning, - body.store === false - ); + if (Array.isArray(body.input)) { + body.input = sanitizeResponsesInput( + body.input, + transport, + incompatibleReasoning || mixedState, + body.store === false + ); + } return { incompatibleReasoning: false }; } diff --git a/open-sse/services/streamRecovery.ts b/open-sse/services/streamRecovery.ts index 37a4867f137..3a95e9a2a3f 100644 --- a/open-sse/services/streamRecovery.ts +++ b/open-sse/services/streamRecovery.ts @@ -183,6 +183,11 @@ export function hasTerminalMarker(bytes: Uint8Array): boolean { export interface OpenAiSseScan { /** Concatenated assistant text seen across `choices[].delta.content`. */ text: string; + /** Concatenated reasoning trace seen across `choices[].delta.reasoning_content`. Some + * providers stream the entire answer here and leave `content` empty/null — tracked + * separately so a clean stop with reasoning-only output can still be recognized as + * "nothing usable was delivered" instead of "a normal empty turn". */ + reasoningText: string; /** True if any `choices[].delta.tool_calls` appeared — NEVER continue those. */ sawToolCall: boolean; /** @@ -201,6 +206,10 @@ export interface OpenAiSseScan { * (and the client-visible SSE) is still eligible to be resumed past it. */ terminal: boolean; + /** The literal `finish_reason` string when present (e.g. "stop", "tool_calls", "length", + * "content_filter"), or `null` if none was seen. `terminal` alone is not precise enough + * to gate the reasoning-only-stop continuation — it must fire on `"stop"` only. */ + finishReason: string | null; /** True if at least one OpenAI-shaped `choices[].delta` was parsed (format gate). */ parsedOpenAi: boolean; } @@ -212,12 +221,22 @@ export interface OpenAiSseScan { */ export function scanOpenAiSseText(sse: string): OpenAiSseScan { let text = ""; + let reasoningText = ""; let sawToolCall = false; let toolCallFinished = false; let terminal = false; + let finishReason: string | null = null; let parsedOpenAi = false; if (typeof sse !== "string" || sse.length === 0) { - return { text, sawToolCall, sawToolCallInFlight: false, terminal, parsedOpenAi }; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight: false, + terminal, + finishReason, + parsedOpenAi, + }; } for (const line of sse.split("\n")) { const trimmed = line.trimStart(); @@ -242,21 +261,33 @@ export function scanOpenAiSseText(sse: string): OpenAiSseScan { parsedOpenAi = true; const content = (delta as { content?: unknown }).content; if (typeof content === "string") text += content; + const reasoning = (delta as { reasoning_content?: unknown }).reasoning_content; + if (typeof reasoning === "string") reasoningText += reasoning; const toolCalls = (delta as { tool_calls?: unknown }).tool_calls; if (Array.isArray(toolCalls) && toolCalls.length > 0) sawToolCall = true; } - const finishReason = (choice as { finish_reason?: unknown })?.finish_reason; - if (finishReason === "tool_calls") { + const rawFinishReason = (choice as { finish_reason?: unknown })?.finish_reason; + if (rawFinishReason === "tool_calls") { // Ends this one choice, but the overall stream/turn stays continuable — // never counts as the general terminal marker (see OpenAiSseScan.terminal). toolCallFinished = true; - } else if (finishReason != null) { + finishReason = "tool_calls"; + } else if (rawFinishReason != null) { terminal = true; + if (typeof rawFinishReason === "string") finishReason = rawFinishReason; } } } const sawToolCallInFlight = sawToolCall && !toolCallFinished; - return { text, sawToolCall, sawToolCallInFlight, terminal, parsedOpenAi }; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight, + terminal, + finishReason, + parsedOpenAi, + }; } export interface ContinuableBody { @@ -267,8 +298,10 @@ export interface ContinuableBody { /** * Build a re-request body that continues from `assistantSoFar` by appending it as an - * assistant turn. Returns null when the body has no `messages` array or the partial text - * is empty (nothing to continue from). Does not mutate the original. + * assistant turn. When `assistantSoFar` is empty (nothing usable was emitted yet — e.g. a + * clean stop that only produced reasoning), the messages are re-sent unchanged instead of + * appending an empty assistant turn: this simply re-asks for a real answer. Returns null + * only when the body has no `messages` array at all (nothing to continue from). */ export function makeContinuationBody( body: ContinuableBody, @@ -276,10 +309,13 @@ export function makeContinuationBody( ): (ContinuableBody & { messages: unknown[] }) | null { if (!body || typeof body !== "object") return null; if (!Array.isArray(body.messages) || body.messages.length === 0) return null; - if (typeof assistantSoFar !== "string" || assistantSoFar.length === 0) return null; + if (typeof assistantSoFar !== "string") return null; return { ...body, - messages: [...body.messages, { role: "assistant", content: assistantSoFar }], + messages: + assistantSoFar.length > 0 + ? [...body.messages, { role: "assistant", content: assistantSoFar }] + : [...body.messages], stream: true, }; } @@ -390,8 +426,13 @@ export function createRecoverableStream( let continuations = 0; let emittedTail = ""; // raw SSE not yet scanned (awaiting an event boundary) let emittedText = ""; // assistant text already delivered to the client + let emittedReasoningText = ""; // reasoning trace already delivered (never shown to the client, + // tracked only to distinguish "a real empty turn" from "the whole + // answer stayed in the reasoning channel") + let emittedFinishReason: string | null = null; // literal finish_reason last seen, if any let emittedTerminal = false; let emittedToolCallInFlight = false; + let emittedSawToolCall = false; // any tool_call delta seen, complete or not let emittedParsedOpenAi = false; // Enqueue to the client and, when continuation is enabled, fold the chunk into the @@ -409,8 +450,11 @@ export function createRecoverableStream( emittedTail = emittedTail.slice(boundary + 2); const scan = scanOpenAiSseText(complete); emittedText += scan.text; + emittedReasoningText += scan.reasoningText; + if (scan.finishReason !== null) emittedFinishReason = scan.finishReason; if (scan.terminal) emittedTerminal = true; if (scan.sawToolCallInFlight) emittedToolCallInFlight = true; + if (scan.sawToolCall) emittedSawToolCall = true; if (scan.parsedOpenAi) emittedParsedOpenAi = true; }; @@ -418,15 +462,42 @@ export function createRecoverableStream( for (const chunk of holdback.flush()) emit(controller, chunk); }; - // A post-commit truncation is continuable only for a plain-text OpenAI-compatible - // stream that has not finished and has no tool call in flight. + // A post-commit truncation is continuable for a plain-text OpenAI-compatible stream that + // has no tool call in flight, AND either: + // - has not finished yet (the original #4131 truncation case), or + // - finished with a literal finish_reason of "stop" but delivered nothing usable while a + // non-empty reasoning trace shows the provider spent its whole turn "thinking" and never + // turned that into an answer (some providers put the entire response in + // reasoning_content and leave content empty). Gated on the LITERAL "stop" value, not the + // generic `terminal` flag — `terminal` also covers "length"/"content_filter"/a bare + // [DONE], which are out of scope for this specific recovery. + // + // Known consequence of the hallucinatedEmptyStop path (flagged in cross-review, accepted as + // inherent to tryContinue's existing design, not new to this fix): the original upstream's + // `finish_reason:"stop"` chunk was already forwarded to the client via `emit()`'s unconditional + // `controller.enqueue(chunk)` (streamRecovery.ts:381) BEFORE this scan ever runs — that is how + // `emittedFinishReason`/`emittedTerminal` get set in the first place. So the client sees an + // empty "stop" marker from the original turn, then — once the continuation succeeds — the real + // answer plus a SECOND `emitCleanTerminal` from `tryContinue`. This mirrors what already + // happens for the pre-existing truncation-continuation case (a truncated stream can likewise + // have partially delivered SSE framing before `tryContinue` appends more); it is not a new + // double-close of the underlying `ReadableStream` (`controller.close()` runs exactly once, + // after `tryContinue` returns). An SSE client that treats a bare `finish_reason:"stop"` as an + // unconditional end-of-turn (rather than waiting for `[DONE]`) may need updating separately — + // out of scope for this fix, which targets the observed opencode/OmniRoute pairing where the + // client kept the connection open. + const hallucinatedEmptyStop = () => + emittedFinishReason === "stop" && + !emittedSawToolCall && + emittedText.length === 0 && + emittedReasoningText.length > 0; + const canContinue = () => continueEnabled && continuations < maxContinuations && emittedParsedOpenAi && !emittedToolCallInFlight && - !emittedTerminal && - emittedText.length > 0; + (emittedText.length > 0 ? !emittedTerminal : hallucinatedEmptyStop()); const emitCleanTerminal = (controller: ReadableStreamDefaultController) => { controller.enqueue( @@ -470,7 +541,24 @@ export function createRecoverableStream( } const scan = scanOpenAiSseText(raw); - const suffix = trimContinuationOverlap(emittedText, scan.text); + // A continuation whose overlap with what was already emitted falls below the documented + // threshold is treated as a suspected restart rather than a real resume — see + // STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS for the full trade-off rationale. This + // is a heuristic, not a proof: it deliberately trades some false-positive rejections of + // legitimate low-overlap continuations against never silently gluing two unrelated + // fragments into one corrupted message. + const overlapResult = trimContinuationOverlap(emittedText, scan.text); + const overlapChars = scan.text.length - overlapResult.length; + const isSuspectedRestart = + emittedText.length > 0 && + scan.text.length > 0 && + overlapChars < STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS; + if (isSuspectedRestart) { + if (await tryContinue(controller)) return true; + emitCleanTerminal(controller); + return true; + } + const suffix = overlapResult; if (suffix) { emit( controller, @@ -527,9 +615,11 @@ export function createRecoverableStream( const { done, value } = result; if (done) { if (holdback.committed) { - // Graceful end after commit: if it lacks a terminal marker it is a silent - // truncation — try to continue; otherwise (clean finish) just close. - if (!emittedTerminal && (await tryContinue(controller))) { + // Graceful end after commit: try a mid-stream continuation whenever canContinue() + // says the stream is worth continuing (silent truncation, or a clean-but-empty + // reasoning-only stop) — canContinue() is the single source of truth here, same as + // the read-error branch above. + if (await tryContinue(controller)) { runFinalize(); controller.close(); return; diff --git a/open-sse/translator/request/openai-responses.ts b/open-sse/translator/request/openai-responses.ts index b8866502524..d48dcdf1816 100644 --- a/open-sse/translator/request/openai-responses.ts +++ b/open-sse/translator/request/openai-responses.ts @@ -8,11 +8,7 @@ import { isOpenAIResponsesStoreEnabled } from "@/lib/providers/requestDefaults"; import { FORMATS } from "../formats.ts"; import { register } from "../registry.ts"; import { normalizeResponsesInputForChat } from "../../utils/responsesInputNormalization.ts"; -import { - createReasoningTransportIncompatibleError, - hasOpaqueReasoningState, - extractReplayableResponsesReasoningText, -} from "../../services/reasoningInputPolicy.ts"; +import { extractReplayableResponsesReasoningText } from "../../services/reasoningInputPolicy.ts"; import { getRegisteredProviders, requiresPlainStringContent, @@ -454,10 +450,8 @@ export function openaiResponsesToOpenAIRequest( if (itemType === "reasoning") { // Only genuine plaintext reasoning can cross into Chat reasoning_content. - // Opaque encrypted state and its display summary have no Chat replay form. - if (preserveReasoningContent && hasOpaqueReasoningState(item)) { - throw createReasoningTransportIncompatibleError(); - } + // Opaque encrypted state and its display summary have no Chat replay form, + // so opaque-only items are dropped while mixed items replay their plaintext. if (preserveReasoningContent) { const reasoning = extractReplayableResponsesReasoningText(item); if (reasoning) { diff --git a/open-sse/utils/proxyFetch.ts b/open-sse/utils/proxyFetch.ts index e8e3ea26f87..4eedd2dad56 100644 --- a/open-sse/utils/proxyFetch.ts +++ b/open-sse/utils/proxyFetch.ts @@ -858,6 +858,12 @@ async function patchedFetch( msg.includes("fetch failed") || errCode === "ECONNREFUSED" || msg.includes("ECONNREFUSED") || + errCode === "EAI_AGAIN" || + msg.includes("EAI_AGAIN") || + errCode === "ENOTFOUND" || + msg.includes("ENOTFOUND") || + errCode === "ETIMEDOUT" || + msg.includes("ETIMEDOUT") || (typeof errCode === "string" && errCode.startsWith("UND_ERR")) || msg.includes("UND_ERR") ) { diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index 4d8993c7670..9adb5dbf2f6 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -683,13 +683,15 @@ export function createDisconnectAwareStream(transformStream, streamController) { if (clientTerminalSeen) return; terminalTail += terminalDecoder.decode(chunk, { stream: true }); - if (terminalTail.length > 4096) { - terminalTail = terminalTail.slice(-4096); - } + // Scan before bounding retained state: a compaction terminal frame can + // exceed the tail budget because encrypted_content is carried inline. clientTerminalSeen = hasClientTerminalSseMarker( terminalTail, streamController.clientResponseFormat ); + if (terminalTail.length > 4096) { + terminalTail = terminalTail.slice(-4096); + } if (clientTerminalSeen) { streamController.markClientTerminalSeen?.(); } diff --git a/open-sse/utils/streamReadiness.ts b/open-sse/utils/streamReadiness.ts index 2d06659b809..1696a7a5c52 100644 --- a/open-sse/utils/streamReadiness.ts +++ b/open-sse/utils/streamReadiness.ts @@ -34,6 +34,14 @@ function hasUsefulValue(value: unknown): boolean { if (Array.isArray(value)) return value.some(hasUsefulValue); if (!isRecord(value)) return false; + // A Responses compaction item IS the turn's output: remote compaction + // completes with output = [{type:"compaction", encrypted_content}] and no + // assistant text. Deliberately NOT a blanket encrypted_content key — an + // encrypted reasoning item alone is not user-visible output and must keep + // tripping the #8649 empty-content guard. + // This shape is specific to Responses streams; chat-completion frames do not produce it. + if (value.type === "compaction" && hasNonEmptyString(value.encrypted_content)) return true; + for (const key of [ "content", "text", diff --git a/open-sse/utils/syncedEffortVariants.ts b/open-sse/utils/syncedEffortVariants.ts index 2a2c7f0d68c..33c5ec8c56b 100644 --- a/open-sse/utils/syncedEffortVariants.ts +++ b/open-sse/utils/syncedEffortVariants.ts @@ -19,17 +19,17 @@ * only when the base model's own `supportedThinkingEfforts` actually declares that tier — * never a blind string match. * - * Skipped entirely for `codex` and `kimi`-owned models: both already own a conflicting - * native `-{effort}` suffix mechanism (`splitCodexReasoningSuffix` / - * `getKimiCodeStaticThinkingPolicy`), so double-registering here would collide with their - * own alias resolution. Also skipped for any model whose id already ends in a token that - * matches a canonical effort value, to avoid colliding with a model that legitimately ends - * in an effort-like token (e.g. a model literally named "...-high"). + * Skipped entirely for `codex`, `kimi`-owned, and GLM (`glm`, `glm-cn`, `glmt`) models: + * they already own conflicting `-{effort}` aliases (`splitCodexReasoningSuffix`, + * `getKimiCodeStaticThinkingPolicy`, or `GlmExecutor::parseGlmEffortTier`), so generating + * another layer here would create invalid nested ids. Also skipped for any model whose id + * already ends in a token that matches a canonical effort value, to avoid colliding with a + * model that legitimately ends in an effort-like token (e.g. a model named "...-high"). */ import { CANONICAL_EFFORT_VALUES } from "@/shared/reasoning/effortStandardization.ts"; -/** Provider ids that already own a native `-{effort}` suffix mechanism — never double-register. */ -export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex"]); +/** Provider ids with dedicated `-{effort}` aliases — never synthesize another suffix layer. */ +export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex", "glm", "glm-cn", "glmt"]); /** Provider-id prefixes covering that mechanism's multiple connection variants (kimi-coding, kimi-coding-apikey). */ const SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES = ["kimi"]; diff --git a/package.json b/package.json index 55cc574f9ec..526c60cbb78 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "omniroute", "version": "3.8.50", - "description": "Unified AI router with 348 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", + "description": "Unified AI router with 349 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { "omniroute": "bin/omniroute.mjs", diff --git a/promise-pillars.svg b/promise-pillars.svg new file mode 100644 index 00000000000..aebefefabbf --- /dev/null +++ b/promise-pillars.svg @@ -0,0 +1,139 @@ + + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. + + + + + + + + + + + + + + + + + + THE PROMISE + + + + One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. + + + + + + + + + + + + + + + + Never hit limits + Auto-fallback across 349 providers in + milliseconds. Quota out? The next provider + takes over — zero downtime. + + + + + + + + + + + + + + + Save up to 95% tokens + RTK + Caveman stacked compression cuts + 15–95% of eligible tokens — ~89% average + on tool-heavy sessions. + + + + + + + + + + + + + + $0 to start + 90+ providers with a free tier, 56 free + forever — Qoder, Pollinations, Cloudflare, + SiliconFlow… No card needed. + + + + + + + + + + + + + + + Every tool works + 33 coding agents — Claude Code, Codex, + Cursor, Cline, Copilot, Antigravity — + through one config. + + + + + + + + + + + + + + One endpoint + OpenAI ↔ Claude ↔ Gemini ↔ Responses API + translation. Point any tool at /v1 — + it just works. + + + + + + + + + + + + + + Production-grade + Circuit breakers, TLS stealth, MCP (110 + tools), A2A, memory, guardrails, evals — + 25,000+ tests. + + + + + + $ npm i -g omniroute  ·  point your tool at http://localhost:20128/v1  ·  $0 + MIT · OPEN SOURCE + + diff --git a/public/providers/logfare.png b/public/providers/logfare.png new file mode 100644 index 00000000000..223f6e39cdc Binary files /dev/null and b/public/providers/logfare.png differ diff --git a/scripts/build/colocateOptionals.mjs b/scripts/build/colocateOptionals.mjs index 0aa3f38fab5..7b03a5c53fc 100644 --- a/scripts/build/colocateOptionals.mjs +++ b/scripts/build/colocateOptionals.mjs @@ -47,7 +47,7 @@ * fail-open, so this never throws into the install. */ -import { cpSync, existsSync, mkdirSync, readFileSync } from "node:fs"; +import { cpSync, existsSync, mkdirSync, readFileSync, realpathSync } from "node:fs"; import { createRequire } from "node:module"; import { dirname, join, sep } from "node:path"; @@ -119,7 +119,9 @@ function isPackageIntact(targetNodeModulesDir, name) { const resolved = probe.resolve(name); // A resolution that walked past the target into an ancestor tree does not // prove the target copy is usable. - return resolved.startsWith(targetNodeModulesDir + sep); + const realTarget = realpathSync(targetNodeModulesDir); + const realResolved = realpathSync(resolved); + return realResolved.startsWith(realTarget + sep); } catch { return false; } diff --git a/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx new file mode 100644 index 00000000000..77cf1932983 --- /dev/null +++ b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx @@ -0,0 +1,108 @@ +"use client"; + +import { useSyncExternalStore } from "react"; +import { useTranslations } from "next-intl"; +import ProviderIcon from "@/shared/components/ProviderIcon"; + +// Branded short link through our own link.omniroute.online shortener, so the +// click lands in our Kutt metrics. Points at cheaperinference.com?utm_source=omniroute +// (the URL in README.md's Open Source Friends section). Keep in sync with the +// `cheaper` slug on the shortener. +const CHEAPER_INFERENCE_URL = "https://link.omniroute.online/cheaper"; + +// Cheaper Inference brand green (#31f889). White text on it fails contrast, so +// the CTA pairs it with the dark ink from the provider's color token (colors.ts: +// cheaperinference.text = #04170d). Hex values stay in sync with that token. + +const DISMISS_STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +// Same-tab signal for the dismiss button, since writing localStorage doesn't +// fire a "storage" event in the tab that wrote it. +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; + +function isNotDismissed(): boolean { + try { + return !localStorage.getItem(DISMISS_STORAGE_KEY); + } catch { + return true; + } +} + +function subscribe(callback: () => void) { + window.addEventListener(DISMISS_EVENT, callback); + return () => window.removeEventListener(DISMISS_EVENT, callback); +} + +// SSR has no localStorage, so the server always renders the banner visible; +// useSyncExternalStore reconciles that against the real client-side value +// right after hydration, mirroring KimiSponsorBanner's pattern. +function getServerSnapshot() { + return true; +} + +/** + * Dismissable banner announcing the Cheaper Inference OmniRoute partnership on + * the dashboard home page — same size/shape as KimiSponsorBanner, no version + * gate (durable partnership, not a time-boxed offer). The logomark reuses + * . + */ +export default function CheaperInferenceSponsorBanner() { + const t = useTranslations("cheaperInferenceSponsorBanner"); + const visible = useSyncExternalStore(subscribe, isNotDismissed, getServerSnapshot); + + if (!visible) { + return null; + } + + const dismiss = () => { + try { + localStorage.setItem(DISMISS_STORAGE_KEY, "true"); + } catch { + // ignore — worst case the banner reappears next visit + } + window.dispatchEvent(new Event(DISMISS_EVENT)); + }; + + return ( +
+
+
+ +
+
+

{t("title")}

+

{t("description")}

+
+
+ +
+
+ + {t("cta")} + + + {t("partnerLinkNote")} +
+ +
+
+ ); +} diff --git a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx index 0b23976531c..1715f17c839 100644 --- a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx +++ b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx @@ -6,7 +6,9 @@ import { useTranslations } from "next-intl"; // Marketplace listing is the primary CTA; Open VSX (Cursor/Windsurf/VSCodium/etc.) // is called out via secondaryNote instead of a second button, to keep this banner // the same size as KimiSponsorBanner. -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +// Branded short link through our own link.omniroute.online shortener (the `vsx` +// slug), so the click lands in our Kutt metrics. +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; const DISMISS_STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; // Same-tab signal for the dismiss button, since writing localStorage doesn't diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 7cb1c6c0eea..04f73d54a0f 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -533,7 +533,7 @@ function getStrategyBadgeClass(strategy) { return "bg-blue-500/15 text-blue-600 dark:text-blue-400"; } -function getI18nOrFallback(t, key, fallback, values) { +function getI18nOrFallback(t, key, fallback, values = undefined) { try { if (typeof t.has === "function" && t.has(key)) return t(key, values); } catch {} @@ -3896,15 +3896,6 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders, combo )} - {config.reasoningTransportFallback !== "skip" && ( -

- {getI18nOrFallback( - t, - "reasoningTransportFallbackDropWarning", - "May lose continuation context or cause tool-call continuations to fail." - )} -

- )}
import("./components/AddCompatibleProviderModal"), + { ssr: false } +); import { CategoryDot } from "./components/CategoryDot"; -import { ImportProvidersFromFileModal } from "./components/ImportProvidersFromFileModal"; +const ImportProvidersFromFileModal = dynamic( + () => + import("./components/ImportProvidersFromFileModal").then( + (m) => m.ImportProvidersFromFileModal + ), + { ssr: false } +); import NoAuthProvidersSection from "./components/NoAuthProvidersSection"; import HighlightableProviderCard from "./components/HighlightableProviderCard"; import ProviderCountBadge from "./components/ProviderCountBadge"; diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx index 3af32da285d..dc102936d09 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx @@ -98,7 +98,7 @@ export function formatQuotaLabel(name: string) { return `Weekly ${toTitleCaseWords(weeklyModelMatch[1])}`; } - return trimmed; + return toTitleCaseWords(trimmed.replace(/_/g, " ")); } /** diff --git a/src/app/(dashboard)/home/page.tsx b/src/app/(dashboard)/home/page.tsx index beccde22610..bc10df88f42 100644 --- a/src/app/(dashboard)/home/page.tsx +++ b/src/app/(dashboard)/home/page.tsx @@ -4,6 +4,7 @@ import { getSettings } from "@/lib/localDb"; import HomePageClient from "../dashboard/HomePageClient"; import BootstrapBanner from "../dashboard/BootstrapBanner"; import KimiSponsorBanner from "../dashboard/KimiSponsorBanner"; +import CheaperInferenceSponsorBanner from "../dashboard/CheaperInferenceSponsorBanner"; import VscodeCopilotBanner from "../dashboard/VscodeCopilotBanner"; import NewsBanner from "../dashboard/NewsBanner"; @@ -20,6 +21,7 @@ export default async function HomePage() { <> {isBootstrapped && } + diff --git a/src/app/api/providers/[id]/models/discovery/providerSets.ts b/src/app/api/providers/[id]/models/discovery/providerSets.ts index 82386814790..2b35e54b01e 100644 --- a/src/app/api/providers/[id]/models/discovery/providerSets.ts +++ b/src/app/api/providers/[id]/models/discovery/providerSets.ts @@ -95,6 +95,11 @@ export const NAMED_OPENAI_STYLE_PROVIDERS = new Set([ "internlm", "ant-ling", "nanogpt", + // Logfare (https://logfare.ai) — free OpenAI-compatible gateway live-verified + // 2026-08-21: GET https://logfare.ai/v1/models returns a real 20-model catalog + // (11 chat-capable). Live fetch keeps it fresh; the registry seed stays as the + // offline fallback. + "logfare", ]); export function isNamedOpenAIStyleProvider(provider: string): boolean { diff --git a/src/app/api/providers/[id]/models/route.ts b/src/app/api/providers/[id]/models/route.ts index 01f1bd72857..cecbb6776eb 100755 --- a/src/app/api/providers/[id]/models/route.ts +++ b/src/app/api/providers/[id]/models/route.ts @@ -1909,12 +1909,9 @@ export async function GET( } if (isAnthropicCompatibleProvider(provider)) { - const cachedResponse = maybeReturnCachedDiscovery(); - if (cachedResponse) return cachedResponse; - - const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled(); - if (autoFetchDisabledResponse) return autoFetchDisabledResponse; - + // CC providers never support models listing — this check must precede + // the cached-discovery / auto-fetch fallbacks, which would otherwise + // return a misleading 200 "no models" for a CC node (#10828 ordering). if (isClaudeCodeCompatibleProvider(provider)) { return NextResponse.json( { error: `Provider ${provider} does not support models listing` }, @@ -1922,6 +1919,12 @@ export async function GET( ); } + const cachedResponse = maybeReturnCachedDiscovery(); + if (cachedResponse) return cachedResponse; + + const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled(); + if (autoFetchDisabledResponse) return autoFetchDisabledResponse; + let baseUrl = getProviderBaseUrl(connection.providerSpecificData); if (!baseUrl) { const fallback = buildDiscoveryFallbackResponse({ diff --git a/src/app/login/page.tsx b/src/app/login/page.tsx index f2c70306f38..664eb3c1721 100644 --- a/src/app/login/page.tsx +++ b/src/app/login/page.tsx @@ -38,8 +38,7 @@ export default function LoginPage() { if (data.nodeVersion) setNodeVersion(data.nodeVersion); if (data.nodeCompatible === false) setNodeCompatible(false); if (data.authenticated === true || data.requireLogin === false) { - router.push("/dashboard"); - router.refresh(); + window.location.href = "/dashboard"; return; } setHasPassword(!!data.hasPassword); @@ -77,13 +76,12 @@ export default function LoginPage() { if (res.ok) { sessionStorage.setItem("omniroute_login_time", String(Date.now())); - router.push("/dashboard"); - router.refresh(); + window.location.href = "/dashboard"; } else { const data = await res.json(); // (#521) If no password is set, redirect to onboarding instead of showing an error if (data.needsSetup) { - router.push("/dashboard/onboarding"); + window.location.href = "/dashboard/onboarding"; return; } setError(data.error || t("invalidPassword")); diff --git a/src/domain/connectionModelRules.ts b/src/domain/connectionModelRules.ts index 316ade72d80..7831bbc84b6 100644 --- a/src/domain/connectionModelRules.ts +++ b/src/domain/connectionModelRules.ts @@ -80,3 +80,26 @@ export function hasEligibleConnectionForModel( (connection) => !isModelExcludedByConnection(modelId, connection?.providerSpecificData) ); } + +/** + * #11089: does this connection's *synced* inventory advertise the model? + * + * Unlike `excludedModels` (a manually maintained denylist) this reads the + * per-connection catalog written by model discovery, so a multi-host local + * provider never routes a model to a host that never had it. Ids are matched + * with the same candidate semantics as the denylist (provider prefix and the + * `[1m]` extended-context suffix are tolerated), but never as wildcard + * patterns — a synced id is a literal. + * + * Fails OPEN on an empty inventory: a host that has not been synced yet is + * "unknown", not "does not have it". + */ +export function isModelAdvertisedByConnection( + modelId: unknown, + advertisedModelIds: ReadonlySet | null | undefined +): boolean { + if (!advertisedModelIds || advertisedModelIds.size === 0) return true; + if (typeof modelId !== "string" || modelId.trim().length === 0) return true; + + return getModelMatchCandidates(modelId).some((candidate) => advertisedModelIds.has(candidate)); +} diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index ff778753c61..930f5b59587 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -13869,5 +13869,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "Cheaper Inference is an OmniRoute Open Source Friend", + "description": "A cost-ranked gateway reselling dozens of frontier models behind one OpenAI-compatible endpoint — routing each request to the cheapest eligible provider, never above list price.", + "cta": "Get an API Key", + "partnerLinkNote": "Partner link", + "dismissAriaLabel": "Dismiss" } } diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index f3e3c15d130..8df20f0f9df 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -4922,7 +4922,9 @@ "multiProvider": "Multi-Provedor", "usageTracking": "Rastreamento de Uso", "securityDesc": "Defina uma senha para proteger seu painel, ou pule por enquanto.", + "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você poderá adicioná-los depois pelo painel, após definir uma senha.", "providerDesc": "Conecte seu primeiro provedor de IA. Você pode adicionar mais depois.", + "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois pelo painel.", "apiKeyRequired": "Chave de API (obrigatório)", "customUrlOptional": "URL personalizada (opcional)", "testDesc": "Vamos verificar se a conexão com seu provedor funciona.", @@ -4979,9 +4981,7 @@ "skipped": "já configurado", "failed": "falhou" } - }, - "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você pode adicioná-los depois no painel após definir uma senha.", - "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois no painel." + } }, "providers": { "title": "Provedores", @@ -6269,6 +6269,20 @@ "webSessionGuideStep3": "Copie a credencial necessária do próprio domínio do provedor. Para cookies, copie apenas o valor do cabeçalho Cookie e omita Cookie:.", "webSessionGuideStep3Manual": "Caminho manual: abra as ferramentas do desenvolvedor do navegador (F12 → Network), atualize a página, abra uma requisição autenticada e copie o valor do cabeçalho Cookie em Request Headers — omita o prefixo Cookie:.", "webSessionGuideStep4": "Cole aqui e verifique a conexão. Se parar de funcionar, faça login novamente e substitua-o por um novo valor.", + "harImportButtonLabel": "Importar arquivo .har", + "harImportButtonBusy": "Importando…", + "harImportButtonHint": "Exporte pela aba Rede das Ferramentas do Desenvolvedor após enviar pelo menos uma mensagem no chat.", + "harImportStatusValid": "Importado — válido por cerca de {minutes} min.", + "harImportStatusExpiringSoon": "Importado — válido por apenas mais cerca de {minutes} min.", + "harImportStatusExpired": "Importado, mas este token expirou há {minutes} min — exporte um HAR novo.", + "harImportStatusUnknownExpiry": "Importado. Não foi possível ler a expiração.", + "harImportErrorNotJson": "Esse arquivo não é um JSON válido — ele é realmente uma exportação .har?", + "harImportErrorNoEntries": "Este HAR não contém entradas de rede.", + "harImportErrorNoChathubUrl": "Nenhuma conexão de chat do Copilot foi encontrada neste HAR. Envie pelo menos uma mensagem em m365.cloud.microsoft antes de exportar.", + "harImportErrorUnparsableUrl": "A conexão de chat foi encontrada, mas não foi possível ler a URL.", + "harImportErrorMissingFields": "A conexão de chat foi encontrada, mas o token estava ausente.", + "harImportErrorReadFailed": "Não foi possível ler esse arquivo.", + "harImportErrorUnknown": "Não foi possível extrair uma credencial desse arquivo HAR.", "webSessionSecurityHint": "Trate isso como uma senha: ela poderá acessar sua conta da web conectada até que ela expire ou seja revogada.", "webNoAuthGuideTitle": "Nenhuma credencial necessária", "webNoAuthGuideBody": "{provider} não precisa de chave de API ou cookie. Salve a conexão para usar seu endpoint web gratuito.", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 11a3606a54c..1921f3fba7f 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -13845,5 +13845,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "A Cheaper Inference é uma Amiga do Código Aberto do OmniRoute", + "description": "Um gateway com custo ordenado que revende dezenas de modelos de fronteira num único endpoint compatível com OpenAI — roteando cada requisição ao provedor elegível mais barato, nunca acima do preço de tabela.", + "cta": "Obter uma Chave de API", + "partnerLinkNote": "Link de parceiro", + "dismissAriaLabel": "Dispensar" } } diff --git a/src/lib/memory/injection.ts b/src/lib/memory/injection.ts index 9fdd2bbbf10..d4d8ead7f7c 100644 --- a/src/lib/memory/injection.ts +++ b/src/lib/memory/injection.ts @@ -65,7 +65,11 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * * Populated with the Xiaomi MiMo endpoint (provider id `xiaomi-mimo`, registry * alias `mimo`, serving mimo-v2.5) confirmed live to 400 on a non-first system - * message. Add other providers here only when they are documented as strict. + * message, and the TokenRouter gateway (provider id `tokenrouter`), confirmed + * live on 2026-08-22 to reject mid-array system messages — including the + * compression notice spliced by purifyHistory() before that splice was fixed to + * merge into the leading system message. Add other providers here only when + * they are documented as strict. * * Self-hosted deployments can extend this list without a source change via * OMNIROUTE_STRICT_SYSTEM_PROVIDERS (comma-separated provider ids, @@ -73,7 +77,7 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * self-hosted Qwen3.5+/3.6 model, whose chat template enforces the same * single-leading-system-message constraint as xiaomi-mimo. */ -const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo"]); +const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo", "tokenrouter"]); /** * Parses OMNIROUTE_STRICT_SYSTEM_PROVIDERS into a normalized id list. diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 3828047b90b..b7aa2aa2164 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -130,6 +130,11 @@ function uniqueStrings(values: Array) { ]; } +export function isGlmFamilyModel(modelId: string, displayName = ""): boolean { + const glmFamilyPattern = /(?:^|[/@:_. -])glm(?=$|[-._ /@:](?:z)?\d|\d)/i; + return glmFamilyPattern.test(modelId) || glmFamilyPattern.test(displayName); +} + function toQualifiedId( providerAlias: string | null, provider: string | null, @@ -477,11 +482,16 @@ export function enrichCatalogModelEntry( ? declaredEffortTiers : sourceDeclaresThinking ? undefined - : extendCodexGpt56EffortValues( - metadata.provider, - metadata.model, - CANONICAL_EFFORT_VALUES - ); + : // #10963: GLM-family models never inherit generic OpenAI tiers — an + // explicit empty list is authoritative unless a provider-declared + // contract exists (handled by declaredEffortTiers above). + isGlmFamilyModel(metadata.model, metadata.displayName) + ? [] + : extendCodexGpt56EffortValues( + metadata.provider, + metadata.model, + CANONICAL_EFFORT_VALUES + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -502,7 +512,9 @@ export function enrichCatalogModelEntry( // #6241: surface thinking support + the canonical effort tiers so the frontend can // render the effort/thinking toggles. `thinking` is kept for back-compat; `supportsThinking` // is the explicit flag and `effort_tiers` lists the selectable reasoning levels - // (only when the model actually supports thinking). + // (only when the model actually supports thinking). An explicit empty registry list + // is authoritative; GLM models also require a provider-declared contract instead of + // inheriting generic OpenAI effort tiers. ...(typeof metadata.capabilities.supportsThinking === "boolean" ? { thinking: metadata.capabilities.supportsThinking, diff --git a/src/lib/proxyLogger.ts b/src/lib/proxyLogger.ts index a9eb4b3805d..8665e20c758 100644 --- a/src/lib/proxyLogger.ts +++ b/src/lib/proxyLogger.ts @@ -194,43 +194,103 @@ export function logProxyEvent(entry: ProxyLogInput) { proxyLogs.length = MAX_IN_MEMORY_ENTRIES; } - // 2. Persist to SQLite + // 2. Queue for background batch persistence (SQLite / Redis) if (shouldPersistToDisk) { + enqueueProxyLog(log); + } + + return log; +} + +// ──────────────── Background Batch Persistence ──────────────── + +const BATCH_FLUSH_INTERVAL_MS = 1000; +const BATCH_SIZE_THRESHOLD = 100; + +let pendingLogsQueue: ProxyLogEntry[] = []; +let batchTimer: NodeJS.Timeout | null = null; + +function ensureBatchTimer() { + if (batchTimer) return; + batchTimer = setInterval(() => { + flushProxyLogsSync(); + }, BATCH_FLUSH_INTERVAL_MS); + if (typeof batchTimer.unref === "function") { + batchTimer.unref(); + } +} + +function enqueueProxyLog(log: ProxyLogEntry) { + pendingLogsQueue.push(log); + ensureBatchTimer(); + if (pendingLogsQueue.length >= BATCH_SIZE_THRESHOLD) { + flushProxyLogsSync(); + } +} + +export function flushProxyLogsSync() { + if (pendingLogsQueue.length === 0) return; + const batch = pendingLogsQueue; + pendingLogsQueue = []; + + // 1. If Redis driver is active, asynchronously publish batch to Redis Stream/Channel + if (process.env.QUOTA_STORE_DRIVER === "redis" || process.env.QUOTA_STORE_REDIS_URL) { try { - const db = getDbInstance(); - db.prepare( - `INSERT INTO proxy_logs (id, timestamp, status, proxy_type, proxy_host, proxy_port, - level, level_id, provider, target_url, public_ip, egress_ip, latency_ms, error, - connection_id, combo_id, account, tls_fingerprint) - VALUES (@id, @timestamp, @status, @proxyType, @proxyHost, @proxyPort, - @level, @levelId, @provider, @targetUrl, @clientIp, @egressIp, @latencyMs, @error, - @connectionId, @comboId, @account, @tlsFingerprint)` - ).run({ - id: log.id, - timestamp: log.timestamp, - status: log.status, - proxyType: log.proxy?.type || null, - proxyHost: log.proxy?.host || null, - proxyPort: log.proxy?.port ? Number(log.proxy.port) : null, - level: log.level, - levelId: log.levelId, - provider: log.provider, - targetUrl: log.targetUrl, - clientIp: log.clientIp, - egressIp: log.egressIp, - latencyMs: log.latencyMs, - error: log.error, - connectionId: log.connectionId, - comboId: log.comboId, - account: log.account, - tlsFingerprint: log.tlsFingerprint ? 1 : 0, - }); - } catch (err: any) { - console.warn("[proxyLogger] Failed to persist:", err.message); + import("@/lib/quota/redisQuotaStore").then(({ getRedisQuotaStore }) => { + const store = getRedisQuotaStore(process.env.QUOTA_STORE_REDIS_URL || ""); + const client = (store as any)?.client; + if (client && typeof client.publish === "function") { + for (const entry of batch) { + client.publish("omniroute:proxy_logs", JSON.stringify(entry)).catch(() => {}); + } + } + }).catch(() => {}); + } catch { + /* ignore redis pub errors */ } } - return log; + // 2. Persist to SQLite using a single transaction for high-performance non-blocking write + try { + const db = getDbInstance(); + const insertStmt = db.prepare( + `INSERT INTO proxy_logs (id, timestamp, status, proxy_type, proxy_host, proxy_port, + level, level_id, provider, target_url, public_ip, egress_ip, latency_ms, error, + connection_id, combo_id, account, tls_fingerprint) + VALUES (@id, @timestamp, @status, @proxyType, @proxyHost, @proxyPort, + @level, @levelId, @provider, @targetUrl, @clientIp, @egressIp, @latencyMs, @error, + @connectionId, @comboId, @account, @tlsFingerprint)` + ); + + const transaction = db.transaction((entries: ProxyLogEntry[]) => { + for (const item of entries) { + insertStmt.run({ + id: item.id, + timestamp: item.timestamp, + status: item.status, + proxyType: item.proxy?.type || null, + proxyHost: item.proxy?.host || null, + proxyPort: item.proxy?.port ? Number(item.proxy.port) : null, + level: item.level, + levelId: item.levelId, + provider: item.provider, + targetUrl: item.targetUrl, + clientIp: item.clientIp, + egressIp: item.egressIp, + latencyMs: item.latencyMs, + error: item.error, + connectionId: item.connectionId, + comboId: item.comboId, + account: item.account, + tlsFingerprint: item.tlsFingerprint ? 1 : 0, + }); + } + }); + + transaction(batch); + } catch (err: any) { + console.warn("[proxyLogger] Failed to write proxy log batch to disk:", err?.message || err); + } } // ──────────────── Query ──────────────── diff --git a/src/lib/services/portProbe.ts b/src/lib/services/portProbe.ts index 2a9fd502bda..9a890f541b4 100644 --- a/src/lib/services/portProbe.ts +++ b/src/lib/services/portProbe.ts @@ -168,11 +168,24 @@ export function parseSsPid(stdout: string): number | null { export function parseNetstatPid(stdout: string, port: number): number | null { for (const line of stdout.split("\n")) { const columns = line.trim().split(/\s+/); - // proto recv-q send-q local-address foreign-address state pid/program + // Linux: proto recv-q send-q local-address foreign-address state pid/program if (columns.length < 7 || columns[5] !== "LISTEN") continue; - if (!columns[3].endsWith(`:${port}`)) continue; - const parsed = Number.parseInt(columns[6], 10); - if (Number.isFinite(parsed)) return parsed; + const linuxAddress = columns[3].endsWith(`:${port}`); + const macAddress = columns[3].endsWith(`.${port}`); + if (!linuxAddress && !macAddress) continue; + + if (linuxAddress) { + const linuxPid = Number.parseInt(columns[6], 10); + if (Number.isFinite(linuxPid)) return linuxPid; + } + + // macOS `netstat -anv -p tcp` appends a `process:pid` column after + // the socket counters. Process names may contain spaces, so scan instead + // of relying on one fixed column index. + for (const column of columns.slice(6)) { + const match = /:(\d+)$/.exec(column); + if (match) return Number.parseInt(match[1], 10); + } } return null; } @@ -197,7 +210,11 @@ const PID_PROBES: ReadonlyArray<{ args: (port) => ["-tlnp", `sport = :${port}`], parse: (stdout) => parseSsPid(stdout), }, - { command: "netstat", args: () => ["-tlnp"], parse: parseNetstatPid }, + { + command: "netstat", + args: () => (process.platform === "darwin" ? ["-anv", "-p", "tcp"] : ["-tlnp"]), + parse: parseNetstatPid, + }, ]; /** Run one probe, resolving null on a missing binary, a non-match or a timeout. */ diff --git a/src/lib/usage/flatRateProviders.ts b/src/lib/usage/flatRateProviders.ts index 8d3eec2c5d7..3454b9b3917 100644 --- a/src/lib/usage/flatRateProviders.ts +++ b/src/lib/usage/flatRateProviders.ts @@ -46,6 +46,11 @@ const FLAT_RATE_SUBSCRIPTION_PROVIDER_IDS: ReadonlySet = new Set([ "glm-cn", // GLM Coding (China) plan "claude", // Claude Code plan (OAuth-only — a Claude Pro/Max subscription) "cc", // Claude Code plan (alias id — same connection, shares the `cc` pricing rows) + // OpenCode Go subscription (https://opencode.ai/go) — a flat monthly fee. It is an + // aggregator reselling GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x, so + // per-token rows price each call at the UNDERLYING model's metered rate and the + // analytics overstatement is large rather than marginal (#11149). + "opencode-go", ]); /** diff --git a/src/lib/usage/internalUsageCommand.ts b/src/lib/usage/internalUsageCommand.ts index 257b8f9c158..37f5f009d1c 100644 --- a/src/lib/usage/internalUsageCommand.ts +++ b/src/lib/usage/internalUsageCommand.ts @@ -13,7 +13,7 @@ const TEXT_PLAIN_HEADERS = { "Content-Type": "text/plain; charset=utf-8" } as co type JsonRecord = Record; -interface UsageCommandApiKeyMetadata { +export interface UsageCommandApiKeyMetadata { id: string; name?: string; allowedConnections?: string[] | null; @@ -31,7 +31,7 @@ interface ProviderConnectionLike { quotaWindowThresholds?: Record | null; } -interface UsageSnapshot { +export interface UsageSnapshot { connectionId: string; provider: string; plan: unknown; @@ -39,7 +39,7 @@ interface UsageSnapshot { quotaWindowThresholds?: Record | null; } -interface UsageCommandSelection { +export interface UsageCommandSelection { preferredProvider?: string | null; preferredConnectionId?: string | null; } @@ -258,7 +258,7 @@ function snapshotFromConnection( }; } -async function collectUsageSnapshots( +export async function collectUsageSnapshots( metadata: UsageCommandApiKeyMetadata, deps: RequiredDeps ): Promise { @@ -525,6 +525,55 @@ function appendQuotaBlock( lines.push(`⏱ reset in ${formatResetIn(getResetAt(match?.quota ?? null), now)}`); } +/** + * Structured form of the usage command — what {@link buildUsageCommandText} + * renders as text, exposed as data for API consumers (the OmniCopilot panel + * asks for it via `?format=json`). Text and JSON share the exact same + * collectors, so the two can never disagree about a number. + * + * The key design constraint is the 403 case: a key without `allowUsageCommand` + * must reach the client as a *structured* reason, not a bare text error — a + * caller rendering a usage panel has to be able to tell "the server does not + * know your limits yet" apart from "this key may not ask". + */ +/** Discriminated so the caller never reads a data field off a refusal: + * `allowed:false` carries only `error`; `allowed:true` carries the data. */ +export type UsageCommandJson = + | { allowed: false; error: { message: string } } + | { + allowed: true; + /** Present only when the key opted into per-key usage limits. */ + personal: unknown | null; + /** The selected provider snapshot, or null when nothing is cached. */ + provider: UsageSnapshot | null; + /** Every connection's snapshot, so a panel can render Codex / Claude / + * OpenCode side by side instead of only the selected one (#11191). The + * single-pick in `provider` is a presentation choice for a terminal; the + * collector already gathered all of them. */ + providers: UsageSnapshot[]; + }; + +export async function buildUsageCommandJson( + metadata: UsageCommandApiKeyMetadata, + deps: InternalUsageCommandDeps = {}, + selection: UsageCommandSelection = {} +): Promise { + const resolvedDeps = await normalizeDeps(deps); + const personal = + metadata.usageLimitEnabled === true + ? await resolvedDeps.getApiKeyUsageLimitStatus( + { + ...metadata, + preferredProvider: selection.preferredProvider ?? metadata.preferredProvider ?? null, + }, + { now: resolvedDeps.now } + ) + : null; + const snapshots = await collectUsageSnapshots(metadata, resolvedDeps); + const provider = selectUsageSnapshot(snapshots, selection); + return { allowed: true, personal, provider, providers: snapshots }; +} + export async function buildUsageCommandText( metadata: UsageCommandApiKeyMetadata, deps: InternalUsageCommandDeps = {}, @@ -588,6 +637,17 @@ function inferHttpUsageCommandSelection(request: Request): UsageCommandSelection } } +/** `?format=json` (or `?format=JSON`) — anything else falls back to the text + * form, which is the historical contract of this endpoint. */ +function wantsUsageCommandJson(request: Request): boolean { + try { + const format = new URL(request.url, "http://localhost").searchParams.get("format"); + return format !== null && format.trim().toLowerCase() === "json"; + } catch { + return false; + } +} + function createPlainUsageCommandResponse(text: string, status = 200): Response { return new Response(text, { status, headers: TEXT_PLAIN_HEADERS }); } @@ -764,22 +824,45 @@ export async function handleInternalUsageCommandHttpRequest( ): Promise { try { const resolvedDeps = await normalizeDeps(deps); + const json = wantsUsageCommandJson(request); const apiKey = extractUsageCommandApiKey(request); if (!apiKey || !(await resolvedDeps.isValidApiKey(apiKey))) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } const metadata = await resolvedDeps.getApiKeyMetadata(apiKey); if (!metadata?.id) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } if (metadata.allowUsageCommand !== true) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_DISABLED_MESSAGE } } satisfies UsageCommandJson, + { status: 403 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_DISABLED_MESSAGE, 403); } + const selection = inferHttpUsageCommandSelection(request); + if (json) { + return Response.json(await buildUsageCommandJson(metadata, resolvedDeps, selection)); + } return createPlainUsageCommandResponse( - await buildUsageCommandText(metadata, resolvedDeps, inferHttpUsageCommandSelection(request)) + await buildUsageCommandText(metadata, resolvedDeps, selection) ); } catch (err) { const body = buildErrorBody(500, err instanceof Error ? err.message : String(err)); diff --git a/src/server/authz/pipeline.ts b/src/server/authz/pipeline.ts index 9f4e46bed1e..d8dacf376d7 100644 --- a/src/server/authz/pipeline.ts +++ b/src/server/authz/pipeline.ts @@ -205,6 +205,7 @@ function drainingResponse(requestId: string): NextResponse { { status: 503 } ); response.headers.set(AUTHZ_HEADER_REQUEST_ID, requestId); + response.headers.set("Retry-After", "5"); return response; } diff --git a/src/shared/components/ProviderIcon.tsx b/src/shared/components/ProviderIcon.tsx index aeb6f777ba0..56d54bf0dc2 100644 --- a/src/shared/components/ProviderIcon.tsx +++ b/src/shared/components/ProviderIcon.tsx @@ -263,6 +263,7 @@ const KNOWN_PNGS = new Set([ "linkup-search", "llamafile", "llamagate", + "logfare", "maritalk", "nanobot", "nanogpt", diff --git a/src/shared/constants/providers.ts b/src/shared/constants/providers.ts index 5c058958e76..b9b2b6951db 100644 --- a/src/shared/constants/providers.ts +++ b/src/shared/constants/providers.ts @@ -141,6 +141,7 @@ export const AGGREGATOR_PROVIDER_IDS = new Set([ "void-ai", "helixmind", "tabitoken", + "logfare", ]); export const ENTERPRISE_CLOUD_PROVIDER_IDS = new Set([ diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts index 2baf13d3aae..6dab614425e 100644 --- a/src/shared/constants/providers/apikey/gateways.ts +++ b/src/shared/constants/providers/apikey/gateways.ts @@ -673,11 +673,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, - freeNote: "Free no-key reverse proxy to Groq (gpt4free project) — rate-limited to 5 req/min.", + hasFree: false, + freeNote: + "No-key reverse proxy to Groq (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-gemini": { id: "g4f-gemini", @@ -687,11 +688,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, - freeNote: "Free no-key reverse proxy to Gemini (gpt4free project) — rate-limited to 5 req/min.", + hasFree: false, + freeNote: + "No-key reverse proxy to Gemini (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-pollinations": { id: "g4f-pollinations", @@ -701,12 +703,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, + hasFree: false, freeNote: - "Free no-key reverse proxy to Pollinations (gpt4free project) — rate-limited to 5 req/min.", + "No-key reverse proxy to Pollinations (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-ollama": { id: "g4f-ollama", @@ -716,11 +718,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, - freeNote: "Free no-key hosted Ollama gateway (gpt4free project) — rate-limited to 5 req/min.", + hasFree: false, + freeNote: + "No-key hosted Ollama gateway (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-nvidia": { id: "g4f-nvidia", @@ -730,12 +733,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, + hasFree: false, freeNote: - "Free no-key reverse proxy to NVIDIA NIM (gpt4free project) — rate-limited to 5 req/min.", + "No-key reverse proxy to NVIDIA NIM (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "vercel-ai-gateway": { id: "vercel-ai-gateway", @@ -1265,6 +1268,29 @@ export const APIKEY_PROVIDERS_GATEWAYS = { apiHint: "Create a helix- key and use https://helixmind.online/v1. OpenAI requests use Bearer authentication; the Anthropic-compatible messages endpoint accepts x-api-key.", }, + // Logfare (https://logfare.ai) — free OpenAI-compatible inference, live-verified + // 2026-08-21 (real /v1/models catalog; 11 chat-capable models incl. kimi-k3, + // deepseek-v4-pro, glm-5.2, gpt-5.6-luna). Key issued instantly at /register + // (username/password, no email). ⚠️ Logfare logs every request in exchange for + // free inference (opt out at /consent) — surfaced in freeNote per the catalog + // convention for data-collecting free providers. + logfare: { + id: "logfare", + alias: "logfare", + name: "Logfare", + icon: "auto_awesome", + color: "#22C55E", + textIcon: "LF", + website: "https://logfare.ai", + hasFree: true, + freeNote: + "Free OpenAI-compatible inference — no rate limits, no card. Logfare logs every request (prompts, completions, metadata) for internal research; opt out at /consent. Read https://logfare.ai/tos and https://logfare.ai/privacy before use.", + authHint: + "Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token.", + apiHint: + "Create a free API key at https://logfare.ai/register, then use https://logfare.ai/v1 as the OpenAI-compatible base URL. Note the request-logging policy: prompts, completions and metadata are logged for research (opt out at https://logfare.ai/consent).", + passthroughModels: true, + }, // TabiToken (https://tabitoken.com) — NewAPI-based Claude gateway. Its public pricing // endpoint lists a Claude-only catalog (Opus 5 / 4.8, each with a -thinking variant), // every model accepting the Anthropic and OpenAI protocols. diff --git a/src/shared/middleware/chatBodyAdmission.ts b/src/shared/middleware/chatBodyAdmission.ts index 1c3c25b904b..9b20d78e5ae 100644 --- a/src/shared/middleware/chatBodyAdmission.ts +++ b/src/shared/middleware/chatBodyAdmission.ts @@ -18,6 +18,7 @@ import { CORS_HEADERS } from "../utils/cors"; import { createHmac } from "crypto"; import v8 from "node:v8"; +import { trackRequest } from "../../lib/gracefulShutdown"; function parsePositiveInt(value: string | undefined, fallback: number): number { const parsed = Number.parseInt(String(value), 10); @@ -229,6 +230,7 @@ export class ChatAdmissionController { tryAcquireHealthyHeadroom(): ChatAdmissionLease | null { if (this.#activeHealthy >= this.healthyHeadroom) return null; this.#activeHealthy += 1; + const done = trackRequest(); let released = false; return { get released() { @@ -238,6 +240,7 @@ export class ChatAdmissionController { if (released) return; released = true; this.#activeHealthy = Math.max(0, this.#activeHealthy - 1); + done(); }, }; } @@ -264,6 +267,7 @@ export class ChatAdmissionController { tryAcquireHeavy(): ChatAdmissionLease | null { if (this.#activeHeavy >= this.maxHeavyInFlight) return null; this.#activeHeavy += 1; + const done = trackRequest(); let released = false; return { get released() { @@ -273,6 +277,7 @@ export class ChatAdmissionController { if (released) return; released = true; this.#activeHeavy = Math.max(0, this.#activeHeavy - 1); + done(); this.#dispatchFair(); }, }; diff --git a/src/shared/validation/schemas/combo.ts b/src/shared/validation/schemas/combo.ts index edeca47e729..db825e11cc9 100644 --- a/src/shared/validation/schemas/combo.ts +++ b/src/shared/validation/schemas/combo.ts @@ -321,7 +321,7 @@ export const createComboSchema = z .object({ name: comboNameSchema, description: z.string().max(2000).optional(), - models: z.array(comboModelEntry).optional().default([]), + models: z.array(comboModelEntry).min(1, "a combo requires at least one model"), strategy: comboStrategySchema.optional().default("priority"), config: comboRuntimeConfigSchema.optional(), allowedProviders: z.array(z.string().trim().min(1).max(200)).max(100).optional(), @@ -380,8 +380,9 @@ export const updateComboSchema = z .object({ name: comboNameSchema.optional(), description: z.string().max(2000).optional().nullable(), - // Creation may leave `models` empty (`omniroute combo create` drafts one - // that way); an update may not, or a working combo loses every target. + // An update may not remove every model from a combo, or a working combo + // loses every target. Creation refuses an empty list too: since the CLI + // gained --models (#10954), an empty draft has no remaining legitimate path. models: z .array(comboModelEntry) .min(1, "an update cannot remove every model from a combo") diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 3ec6e6c6d6c..aadf7d29d3d 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -1851,7 +1851,7 @@ async function handleSingleModelChat( modelPinned: runtimeOptions?.modelPinned ?? false, routingComboId: runtimeOptions?.routingComboId ?? null, sessionAffinityKey: runtimeOptions.sessionAffinityKey ?? null, - reasoningTransportFallback: runtimeOptions.reasoningTransportFallback ?? "skip", + reasoningTransportFallback: runtimeOptions.reasoningTransportFallback ?? "drop", managedLease: runtimeOptions.managedLease ?? null, }, runtimeOptions @@ -2240,6 +2240,7 @@ async function handleSingleModelChat( if ( !runtimeOptions.emergencyFallbackTried && !comboName && + !forceLiveComboTest && shouldRetrySameAccountTransport({ status: result.status, errorText: errorStr, diff --git a/src/sse/handlers/chatHelpers.ts b/src/sse/handlers/chatHelpers.ts index 36bd4077408..bcf9a51c62c 100644 --- a/src/sse/handlers/chatHelpers.ts +++ b/src/sse/handlers/chatHelpers.ts @@ -422,7 +422,7 @@ export async function executeChatWithBreaker({ conversationId = null, modelPinned = false, routingComboId = null, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", sessionAffinityKey = null, managedLease = null, }: ExecuteChatWithBreakerOptions): Promise { diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index b1df29552a4..5d3d8851786 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -106,8 +106,17 @@ import { resolveProviderId, NOAUTH_PROVIDERS, WEB_COOKIE_PROVIDERS, + isSelfHostedChatProvider, } from "@/shared/constants/providers"; -import { isModelExcludedByConnection } from "@/domain/connectionModelRules"; +import { + isModelExcludedByConnection, + isModelAdvertisedByConnection, +} from "@/domain/connectionModelRules"; +import { + getSyncedAvailableModelsByConnection, + SYNCED_AVAILABLE_MODELS_MALFORMED, + type SyncedAvailableModelsByConnection, +} from "@/lib/db/models"; import { isFreeModel } from "@/shared/utils/freeModels"; import { applySessionAffinityPin, @@ -1160,6 +1169,54 @@ function materializeConnection( }; } +/** + * #11089: load the per-connection synced model inventory for self-hosted chat + * providers so connection selection can drop hosts that never advertised the + * requested model. + * + * Scoped to SELF_HOSTED_CHAT_PROVIDER_IDS: those are the providers where one + * provider id fans out to several independent hosts with genuinely different + * inventories. Hosted providers share one catalog per provider, so filtering + * there would only add a DB read. + * + * Returns an empty map (= no filtering) when there is no model to match, when + * no candidate is self-hosted, or when the persisted rows are malformed — a + * partial read must never silently shrink the pool. + */ +async function loadAdvertisedModelsForSelfHostedConnections( + connections: ProviderConnectionView[], + requestedModel: string | null +): Promise>> { + const advertised = new Map>(); + if (!requestedModel) return advertised; + + const selfHostedProviders = new Set( + connections + .map((c) => c.provider) + .filter((p): p is string => typeof p === "string" && isSelfHostedChatProvider(p)) + ); + if (selfHostedProviders.size === 0) return advertised; + + await Promise.all( + [...selfHostedProviders].map(async (providerId) => { + let byConnection: SyncedAvailableModelsByConnection; + try { + byConnection = await getSyncedAvailableModelsByConnection(providerId); + } catch { + return; + } + // Malformed persisted rows: fail open for the whole provider. + if (byConnection[SYNCED_AVAILABLE_MODELS_MALFORMED]) return; + for (const [connectionId, models] of Object.entries(byConnection)) { + if (!Array.isArray(models) || models.length === 0) continue; + advertised.set(connectionId, new Set(models.map((m) => m.id))); + } + }) + ); + + return advertised; +} + /** * Get provider credentials from localDb * Filters out unavailable accounts and returns the selected account based on strategy @@ -1435,6 +1492,14 @@ export async function getProviderCredentials( let modelLockedCount = 0; let familyLockedCount = 0; const connectionFilterStatus = new Map(); + // #11089: multi-host self-hosted providers keep a per-connection synced + // inventory. Without it, a request can be routed to a host that never had + // the model, producing a spurious model-not-found instead of pinning to + // the host that does. Empty map = no inventory known = no filtering. + const advertisedModelsByConnection = await loadAdvertisedModelsForSelfHostedConnections( + connections, + requestedModel + ); // Filter out unavailable accounts and excluded connection let availableConnections = connections.filter((c) => { if (excludedConnectionIds.has(c.id)) { @@ -1445,6 +1510,13 @@ export async function getProviderCredentials( connectionFilterStatus.set(c.id, "modelExcluded"); return false; } + if ( + requestedModel && + !isModelAdvertisedByConnection(requestedModel, advertisedModelsByConnection.get(c.id)) + ) { + connectionFilterStatus.set(c.id, "modelNotAdvertised"); + return false; + } if (!allowSuppressedConnections) { if (!allowRateLimitedConnections && isAccountUnavailable(c.rateLimitedUntil)) { connectionFilterStatus.set(c.id, "rateLimited"); @@ -1510,6 +1582,7 @@ export async function getProviderCredentials( const codexScopeLimited = status === "codexScopeLimited"; const modelLocked = status === "modelLocked"; const modelExcluded = status === "modelExcluded"; + const modelNotAdvertised = status === "modelNotAdvertised"; if (excluded || rateLimited) { log.debug( "AUTH", @@ -1520,6 +1593,11 @@ export async function getProviderCredentials( "AUTH", ` → ${c.id?.slice(0, 8)} | excluded by per-account model rule for ${requestedModel}` ); + } else if (modelNotAdvertised) { + log.debug( + "AUTH", + ` → ${c.id?.slice(0, 8)} | synced inventory does not advertise ${requestedModel}` + ); } else if (terminalStatus) { log.debug( "AUTH", @@ -3104,11 +3182,9 @@ export async function clearAccountError( } /** - * Optional CAS token. When provided, the clear is performed via an atomic - * conditional UPDATE (clearConnectionErrorIfUnchanged) that aborts if the row - * was written by a concurrent path between the caller's snapshot read and this - * clear. Closes the TOCTOU window in the quota-recovery path. When omitted, - * the clear is unconditional (preserves existing post-success-call behavior). + * Optional CAS token. When provided, clearConnectionErrorIfUnchanged atomically + * aborts if another path modified the row after the caller's snapshot. + * This closes the TOCTOU window; omission preserves unconditional clearing. */ export interface RecoveredStateExpectation { testStatus: string | null; @@ -3116,23 +3192,25 @@ export interface RecoveredStateExpectation { rateLimitedUntil: string | null; } export async function clearRecoveredProviderState( - credentials: Partial | null, + credentials: unknown, expectedState?: RecoveredStateExpectation ): Promise<{ applied: boolean }> { - if (!credentials?.connectionId) return { applied: false }; + const recoverable = credentials as Partial | null; + if (typeof recoverable?.connectionId !== "string" || !recoverable.connectionId) + return { applied: false }; if (expectedState) { - const applied = await clearConnectionErrorIfUnchanged(credentials.connectionId, expectedState); + const applied = await clearConnectionErrorIfUnchanged(recoverable.connectionId, expectedState); if (!applied) { log.info( "AUTH", - `Skipped recovery clear for ${credentials.connectionId.slice(0, 8)} — state changed concurrently (CAS miss)` + `Skipped recovery clear for ${recoverable.connectionId.slice(0, 8)} — state changed concurrently (CAS miss)` ); return { applied: false }; } - log.info("AUTH", `Account ${credentials.connectionId.slice(0, 8)} error cleared (CAS)`); + log.info("AUTH", `Account ${recoverable.connectionId.slice(0, 8)} error cleared (CAS)`); return { applied: true }; } - await clearAccountError(credentials.connectionId, credentials); + await clearAccountError(recoverable.connectionId, recoverable); return { applied: true }; } type AuthRequestLike = { diff --git a/tests/snapshots/provider/translate-path.json b/tests/snapshots/provider/translate-path.json index e371bbc8581..7a6b93390a0 100644 --- a/tests/snapshots/provider/translate-path.json +++ b/tests/snapshots/provider/translate-path.json @@ -3585,6 +3585,29 @@ "stream": "https://arena.ai/nextjs-api/stream/create-evaluation" } }, + "logfare": { + "format": "openai", + "headers": { + "apiKey": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "nonStream": { + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "oauth": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + } + }, + "url": { + "nonStream": "https://logfare.ai/v1/chat/completions", + "stream": "https://logfare.ai/v1/chat/completions" + } + }, "longcat": { "format": "openai", "headers": { diff --git a/tests/unit/account-rotation-lot-c.test.ts b/tests/unit/account-rotation-lot-c.test.ts index 0dde306f90a..6ae41c42ed8 100644 --- a/tests/unit/account-rotation-lot-c.test.ts +++ b/tests/unit/account-rotation-lot-c.test.ts @@ -1,6 +1,11 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { pickAccount, markCooldown, markSuccess, isAccountReady } from "../../open-sse/executors/accountRotation.ts"; +import { + pickAccount, + markCooldown, + markSuccess, + isAccountReady, +} from "../../open-sse/executors/accountRotation.ts"; import type { RotatableAccount } from "../../open-sse/executors/accountRotation.ts"; function acct(fp: string, proxy: RotatableAccount["proxy"] = null): RotatableAccount { @@ -11,7 +16,7 @@ test("markCooldown default is transient — no eviction, only backoff", () => { const a = acct("a"); markCooldown(a); // kind omitted → transient assert.ok(a.cooldownUntil > Date.now()); - assert.equal((a as Record).evictedAt, undefined); + assert.equal(a.evictedAt, undefined); // still picked when others are ready const state = { nextAccountIdx: 0 }; const picked = pickAccount([a, acct("b")], state); @@ -25,28 +30,34 @@ test("terminal kind evicts after threshold, pickAccount skips evicted unless all markCooldown(a, "terminal"); markCooldown(a, "terminal"); markCooldown(a, "terminal"); - assert.ok((a as Record).evictedAt != null); + assert.ok(a.evictedAt != null); const state = { nextAccountIdx: 0 }; // b is ready, a evicted → b is picked - const picked = pickAccount([a, b], state, (x) => isAccountReady(x) && !(x as Record).evictedAt); + const picked = pickAccount([a, b], state, (x) => isAccountReady(x) && !x.evictedAt); assert.equal(picked.fingerprint, "healthy"); // when all evicted, caller still gets an account rather than hanging (preserves :52-58) - (b as Record).evictedAt = Date.now(); - const fallback = pickAccount([a, b], { nextAccountIdx: 0 }, (x) => isAccountReady(x) && !(x as Record).evictedAt); + b.evictedAt = Date.now(); + const fallback = pickAccount( + [a, b], + { nextAccountIdx: 0 }, + (x) => isAccountReady(x) && !x.evictedAt + ); assert.ok(fallback.fingerprint === "dead" || fallback.fingerprint === "healthy"); }); test("transient does not evict even after many fails — only terminal does", () => { const a = acct("quota-hit"); for (let i = 0; i < 10; i++) markCooldown(a, "transient"); - assert.equal((a as Record).evictedAt, undefined); + assert.equal(a.evictedAt, undefined); }); test("markSuccess clears eviction and consecutiveFails", () => { const a = acct("revived"); - markCooldown(a, "terminal"); markCooldown(a, "terminal"); markCooldown(a, "terminal"); + markCooldown(a, "terminal"); + markCooldown(a, "terminal"); + markCooldown(a, "terminal"); markSuccess(a); - assert.equal((a as Record).evictedAt, null); + assert.equal(a.evictedAt, null); assert.equal(a.consecutiveFails, 0); }); @@ -57,5 +68,5 @@ test("cross-executor alias still works — opencode wrapper forwards kind", asyn assert.ok(mc.length >= 1 && mc.length <= 2); // Prove it accepts terminal without throw const tmp = acct("probe"); - assert.doesNotThrow(() => (mc as Record)(tmp, "terminal")); + assert.doesNotThrow(() => mc(tmp, "terminal")); }); diff --git a/tests/unit/account-rotation.test.ts b/tests/unit/account-rotation.test.ts index f4864ee2dd8..a682f02a0d8 100644 --- a/tests/unit/account-rotation.test.ts +++ b/tests/unit/account-rotation.test.ts @@ -7,6 +7,8 @@ import { markSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, type RotatableAccount, } from "../../open-sse/executors/accountRotation.ts"; @@ -114,3 +116,100 @@ describe("accountRotation", () => { assert.strictEqual(isNetworkErrorRotatable(withoutProxy), false); }); }); + +describe("isEmptyUpstreamRejection", () => { + it("matches the observed malformed completion envelope (no error field, empty content, null finish_reason)", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(isEmptyUpstreamRejection(400, observed), true); + }); + + it("does not match a non-400 status", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(isEmptyUpstreamRejection(200, observed), false); + assert.strictEqual(isEmptyUpstreamRejection(429, observed), false); + assert.strictEqual(isEmptyUpstreamRejection(502, observed), false); + }); + + it("does not match when an error field is present", () => { + const withError = JSON.stringify({ + error: { message: "bad request", type: "invalid_request_error" }, + }); + assert.strictEqual(isEmptyUpstreamRejection(400, withError), false); + const emptyError = JSON.stringify({ error: {} }); + assert.strictEqual(isEmptyUpstreamRejection(400, emptyError), false); + }); + + it("does not match when content is non-empty or tool_calls present", () => { + const nonEmpty = JSON.stringify({ + choices: [{ message: { role: "assistant", content: "hi" }, finish_reason: "stop" }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, nonEmpty), false); + + const toolCalls = JSON.stringify({ + choices: [ + { message: { role: "assistant", tool_calls: [{ id: "x" }] }, finish_reason: "tool_calls" }, + ], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, toolCalls), false); + }); + + it("does not match when content is a non-string non-null value (number, block array)", () => { + const numericContent = JSON.stringify({ + choices: [{ message: { role: "assistant", content: 123 }, finish_reason: null }], + }); + assert.strictEqual( + isEmptyUpstreamRejection(400, numericContent), + false, + "non-string non-null content is not eligible" + ); + + const reasoningContent = JSON.stringify({ + choices: [ + { message: { role: "assistant", reasoning_content: "thinking" }, finish_reason: null }, + ], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, reasoningContent), false); + }); + + it("does not match when choices or message are absent", () => { + const noChoices = JSON.stringify({ id: "chatcmpl_x", model: "muse" }); + assert.strictEqual(isEmptyUpstreamRejection(400, noChoices), false); + const noMessage = JSON.stringify({ choices: [{ finish_reason: null }] }); + assert.strictEqual(isEmptyUpstreamRejection(400, noMessage), false); + }); + + it("does not match when finish_reason is a literal value (not null)", () => { + const stopReason = JSON.stringify({ + choices: [{ message: { role: "assistant" }, finish_reason: "stop" }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, stopReason), false); + }); + + it("matches an empty string content (treated as eligible)", () => { + const emptyContent = JSON.stringify({ + choices: [{ message: { role: "assistant", content: "" }, finish_reason: null }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, emptyContent), true); + }); + + it("returns false for unparseable JSON rather than throwing", () => { + assert.strictEqual(isEmptyUpstreamRejection(400, "not json"), false); + assert.strictEqual(isEmptyUpstreamRejection(400, ""), false); + }); +}); + +describe("extractChatcmplId", () => { + it("extracts the chatcmpl id from an observed envelope", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(extractChatcmplId(observed), "chatcmpl_44fn2g6e7kk"); + }); + + it("falls back to 'unknown' when no id is present", () => { + assert.strictEqual(extractChatcmplId("{choices:[]}"), "unknown"); + assert.strictEqual(extractChatcmplId(""), "unknown"); + assert.strictEqual(extractChatcmplId("not json"), "unknown"); + }); +}); diff --git a/tests/unit/antigravity-dynamic-session-id-10443.test.ts b/tests/unit/antigravity-dynamic-session-id-10443.test.ts new file mode 100644 index 00000000000..af5d198ff9b --- /dev/null +++ b/tests/unit/antigravity-dynamic-session-id-10443.test.ts @@ -0,0 +1,18 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { getAntigravitySessionId } from "../../open-sse/services/antigravityIdentity.ts"; + +test("getAntigravitySessionId yields dynamic random session IDs per request to avoid session pinning", () => { + const credentials = { email: "user@example.com", connectionId: "conn_123" }; + + const id1 = getAntigravitySessionId(credentials); + const id2 = getAntigravitySessionId(credentials); + + assert.notEqual(id1, id2, "getAntigravitySessionId should not pin to a static account email hash"); + assert.equal(typeof id1, "string"); + assert.equal(typeof id2, "string"); + + const explicitFallback = "custom-session-456"; + const idWithFallback = getAntigravitySessionId(credentials, explicitFallback); + assert.equal(idWithFallback, explicitFallback, "explicit fallback session ID should take precedence"); +}); diff --git a/tests/unit/auth-clear-account-error.test.ts b/tests/unit/auth-clear-account-error.test.ts index 1131ae63701..a303da4e7ed 100644 --- a/tests/unit/auth-clear-account-error.test.ts +++ b/tests/unit/auth-clear-account-error.test.ts @@ -104,6 +104,11 @@ test("clearRecoveredProviderState ignores empty payloads and clears recoverable await auth.clearRecoveredProviderState(null); await auth.clearRecoveredProviderState({}); + await auth.clearRecoveredProviderState({ + allExpired: true, + expiredCount: 1, + expiredStatus: "expired", + }); await auth.clearRecoveredProviderState({ connectionId: created.id, testStatus: "unavailable", diff --git a/tests/unit/authz/pipeline.test.ts b/tests/unit/authz/pipeline.test.ts index 34703f270ec..6f6469e804e 100644 --- a/tests/unit/authz/pipeline.test.ts +++ b/tests/unit/authz/pipeline.test.ts @@ -306,6 +306,7 @@ test("runAuthzPipeline rejects new API requests during shutdown drain", async () assert.equal(response.status, 503); assert.equal(body.error.code, "SERVICE_UNAVAILABLE"); + assert.equal(response.headers.get("retry-after"), "5"); }); test("runAuthzPipeline rejects rewritten API aliases during shutdown drain", async () => { @@ -319,6 +320,7 @@ test("runAuthzPipeline rejects rewritten API aliases during shutdown drain", asy assert.equal(response.status, 503); assert.equal(response.headers.get("x-omniroute-route-class"), "CLIENT_API"); assert.equal(body.error.code, "SERVICE_UNAVAILABLE"); + assert.equal(response.headers.get("retry-after"), "5"); }); test("runAuthzPipeline allows dashboard sessions to read model catalog aliases", async () => { diff --git a/tests/unit/auto-keyless-custom-provider-11180.test.ts b/tests/unit/auto-keyless-custom-provider-11180.test.ts new file mode 100644 index 00000000000..c2e97e8158d --- /dev/null +++ b/tests/unit/auto-keyless-custom-provider-11180.test.ts @@ -0,0 +1,91 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// #11180 regression guard: a custom OpenAI-compatible connection pointing at a +// keyless local backend (llama.cpp / Ollama / vLLM started without an API key) +// carries no apiKey, no OAuth token and no provider-specific session data, so +// `hasUsableConnectionCredential` dropped it from `validConnections` before the +// auto/* candidate pool was built. The connection was active, tested and synced, +// yet structurally invisible to auto-routing with no log line and no UI hint. +// +// Keyless is the NORMAL configuration for a self-hosted backend, so a custom +// compatible connection must stay eligible. This gate is one step later than +// #5873 (registry-absent defaultModel fallback), whose guard still passes. + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-auto-keyless-11180-")); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; + +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const virtualFactory = await import("../../open-sse/services/autoCombo/virtualFactory.ts"); + +type VirtualComboResult = Awaited>; + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + await resetStorage(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + + if (ORIGINAL_DATA_DIR === undefined) { + delete process.env.DATA_DIR; + } else { + process.env.DATA_DIR = ORIGINAL_DATA_DIR; + } +}); + +test("keyless custom openai-compatible connection enters the auto pool (#11180)", async () => { + const customProvider = "openai-compatible-chat-c2fe8a44-f2fd-47b4-8893-6f1521804c45"; + await providersDb.createProviderConnection({ + provider: customProvider, + authType: "apikey", + name: "llamaAsimov", + // Keyless local backend: llama-server --host 0.0.0.0 with no --api-key. + apiKey: "", + defaultModel: "Qwen3.8-27B-UD-Q4-DFlash-GGUF", + }); + + const combo: VirtualComboResult = await virtualFactory.createVirtualAutoCombo("fast"); + + const candidate = combo.models.find((model) => model.providerId === customProvider); + assert.ok( + candidate, + "a keyless custom-compatible connection must not be dropped by the credential gate" + ); + assert.equal(candidate.model, `${customProvider}/Qwen3.8-27B-UD-Q4-DFlash-GGUF`); + assert.ok(combo.autoConfig.candidatePool.includes(customProvider)); +}); + +test("a keyless FIRST-PARTY provider connection stays out of the pool (#11180)", async () => { + // The relaxation is scoped to custom compatible connection IDs. A first-party + // provider with an empty key is an unconfigured connection, not a keyless + // local backend, and must still be filtered out. + await providersDb.createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "unconfigured openai", + apiKey: "", + defaultModel: "gpt-4o", + }); + + const combo: VirtualComboResult = await virtualFactory.createVirtualAutoCombo("fast"); + + assert.equal( + combo.autoConfig.candidatePool.includes("openai"), + false, + "an unconfigured first-party connection must remain excluded" + ); +}); diff --git a/tests/unit/capture-critical-db-state.test.ts b/tests/unit/capture-critical-db-state.test.ts index fcf2a017b7e..957c1afce36 100644 --- a/tests/unit/capture-critical-db-state.test.ts +++ b/tests/unit/capture-critical-db-state.test.ts @@ -4,6 +4,8 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +type CoreModule = typeof import("../../src/lib/db/core.ts"); + // Shared across all tests — the module caches DATA_DIR / SQLITE_FILE at load time, // so we must create the temp dir and import exactly once. type CoreModule = typeof import("../../src/lib/db/core.ts"); diff --git a/tests/unit/cc-compatible-provider.test.ts b/tests/unit/cc-compatible-provider.test.ts index 9784382ca38..cbb16a28194 100644 --- a/tests/unit/cc-compatible-provider.test.ts +++ b/tests/unit/cc-compatible-provider.test.ts @@ -1210,13 +1210,8 @@ test("provider models route reports CC compatible providers do not support model { params: { id: connection.id } } ); - assert.ok( - response.status === 400 || response.status === 200, - `CC-compatible models route should 400 (unsupported) or 200 (listed), got ${response.status}` - ); - if (response.status === 400) { - assert.deepEqual(await response.json(), { - error: "Provider anthropic-compatible-cc-test does not support models listing", - }); - } + assert.equal(response.status, 400); + assert.deepEqual(await response.json(), { + error: "Provider anthropic-compatible-cc-test does not support models listing", + }); }); diff --git a/tests/unit/chat-body-admission.test.ts b/tests/unit/chat-body-admission.test.ts index e4707402a15..05afae869c8 100644 --- a/tests/unit/chat-body-admission.test.ts +++ b/tests/unit/chat-body-admission.test.ts @@ -16,6 +16,7 @@ const { resolveSelfLoopBearer, } = admissionModule; const { withEarlyStreamKeepalive } = await import("../../open-sse/utils/earlyStreamKeepalive.ts"); +const { getActiveRequestCount } = await import("../../src/lib/gracefulShutdown.ts"); /** * Save/restore the env-var keys that `resolveSelfLoopBearer` reads so tests can @@ -49,6 +50,25 @@ function chatRequest(body: string, contentLength: string | null = String(body.le }); } +test("heavyweight leases are counted for SIGTERM drain (#11015)", () => { + globalThis.__omnirouteShutdown = { init: true, shuttingDown: false, activeRequests: 0 }; + const controller = new ChatAdmissionController(2); + const before = getActiveRequestCount(); + const lease = controller.tryAcquireHeavy(); + assert.ok(lease); + assert.equal(getActiveRequestCount(), before + 1); + const headroom = controller.tryAcquireHealthyHeadroom(); + assert.ok(headroom); + assert.equal(getActiveRequestCount(), before + 2); + lease.release(); + assert.equal(getActiveRequestCount(), before + 1); + headroom.release(); + assert.equal(getActiveRequestCount(), before); + lease.release(); + headroom.release(); + assert.equal(getActiveRequestCount(), before); +}); + test("small known body is admitted without consuming heavyweight capacity", async () => { const controller = new ChatAdmissionController(1); const result = await admitChatRequest(chatRequest("{}"), { diff --git a/tests/unit/chat-routing-synced-inventory-11089.test.ts b/tests/unit/chat-routing-synced-inventory-11089.test.ts new file mode 100644 index 00000000000..441b14893b6 --- /dev/null +++ b/tests/unit/chat-routing-synced-inventory-11089.test.ts @@ -0,0 +1,184 @@ +/** + * tests/unit/chat-routing-synced-inventory-11089.test.ts + * + * #11089 — Chat routing ignores per-connection model inventory on multi-host + * local providers. + * + * One self-hosted provider (`ollama-local`) with TWO connections pointing at + * different hosts and DISJOINT synced inventories: + * + * studio (priority 1) → gemma3:4b, flux2-klein:9b + * jetson (priority 2) → gemma3:4b + * + * `getProviderCredentials` only ever consulted the manual `excludedModels` + * denylist, never the synced inventory persisted per connection, so a request + * for `flux2-klein:9b` could land on jetson — a host that never had the model. + * + * Cases: + * 1. Model advertised by only one connection → the other is never selected. + * 2. Higher-priority host cooling → must NOT preemptively fall to a host that + * lacks the model. + * 3. The advertising connection stays selectable. + * 4. Model advertised by both → both remain eligible (no over-filtering). + * 5. Provider with NO synced inventory at all → fail open, selection unchanged. + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-chat-synced-11089-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const modelsDb = await import("../../src/lib/db/models.ts"); +const auth = await import("../../src/sse/services/auth.ts"); + +const PROVIDER = "ollama-local"; + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +async function createConnection(data: Record): Promise { + const created = (await providersDb.createProviderConnection(data)) as { id: string }; + return created.id; +} + +/** The connection id the selector handed back, or null if it returned no account. */ +function selectedConnectionId(selected: unknown): string | null { + if (!selected || typeof selected !== "object") return null; + const id = (selected as { connectionId?: unknown }).connectionId; + return typeof id === "string" ? id : null; +} + +/** Create the two-host ollama-local topology from the issue report. */ +async function seedTwoHosts(options: { studioRateLimitedUntil?: string } = {}) { + const studioId = await createConnection({ + provider: PROVIDER, + authType: "none", + name: "Mac Studio", + baseUrl: "http://studio.lan:11434/v1", + priority: 1, + isActive: true, + }); + const jetsonId = await createConnection({ + provider: PROVIDER, + authType: "none", + name: "Jetson", + baseUrl: "http://jetson.lan:11434/v1", + priority: 2, + isActive: true, + }); + + await modelsDb.replaceSyncedAvailableModelsForConnection(PROVIDER, studioId, [ + { id: "gemma3:4b", name: "gemma3:4b" }, + { id: "flux2-klein:9b", name: "flux2-klein:9b" }, + ]); + await modelsDb.replaceSyncedAvailableModelsForConnection(PROVIDER, jetsonId, [ + { id: "gemma3:4b", name: "gemma3:4b" }, + ]); + + if (options.studioRateLimitedUntil) { + await providersDb.updateProviderConnection(studioId, { + rateLimitedUntil: options.studioRateLimitedUntil, + }); + } + + return { studioId, jetsonId }; +} + +test("#11089 selects only the host whose synced inventory advertises the model", async () => { + await resetStorage(); + const { studioId, jetsonId } = await seedTwoHosts(); + + // Exclude studio to force the selector to look elsewhere. Jetson does not + // advertise flux2-klein:9b, so it must NOT be handed back. + const selected = await auth.getProviderCredentials(PROVIDER, studioId, null, "flux2-klein:9b"); + + assert.notEqual( + selectedConnectionId(selected), + jetsonId, + "jetson never synced flux2-klein:9b and must not be selected for it" + ); +}); + +test("#11089 does not preemptively fail over to a host lacking the model when the owner is cooling", async () => { + await resetStorage(); + const coolingUntil = new Date(Date.now() + 10 * 60 * 1000).toISOString(); + const { jetsonId } = await seedTwoHosts({ studioRateLimitedUntil: coolingUntil }); + + const selected = await auth.getProviderCredentials(PROVIDER, null, null, "flux2-klein:9b"); + + assert.notEqual( + selectedConnectionId(selected), + jetsonId, + "a cooling studio must surface a cooldown, not silently route to a host without the model" + ); +}); + +test("#11089 keeps the connection that does advertise the model selectable", async () => { + await resetStorage(); + const { studioId } = await seedTwoHosts(); + + const selected = await auth.getProviderCredentials(PROVIDER, null, null, "flux2-klein:9b"); + + assert.equal( + selectedConnectionId(selected), + studioId, + "studio advertises flux2-klein:9b and must be selected" + ); +}); + +test("#11089 a model advertised by every host leaves both connections eligible", async () => { + await resetStorage(); + const { studioId, jetsonId } = await seedTwoHosts(); + + const first = await auth.getProviderCredentials(PROVIDER, null, null, "gemma3:4b"); + assert.equal( + selectedConnectionId(first), + studioId, + "fill-first prefers priority 1 for a shared model" + ); + + // Excluding studio (the normal account-fallback path) must still reach jetson, + // because jetson genuinely advertises gemma3:4b. + const second = await auth.getProviderCredentials(PROVIDER, studioId, null, "gemma3:4b"); + assert.equal( + selectedConnectionId(second), + jetsonId, + "jetson advertises gemma3:4b and must remain a valid failover" + ); +}); + +test("#11089 fails open when the provider has no synced inventory at all", async () => { + await resetStorage(); + + const connectionId = await createConnection({ + provider: PROVIDER, + authType: "none", + baseUrl: "http://127.0.0.1:11434/v1", + priority: 1, + isActive: true, + }); + + // No replaceSyncedAvailableModelsForConnection call: discovery never ran. + // Routing must behave exactly as before rather than filtering everything out. + const selected = await auth.getProviderCredentials(PROVIDER, null, null, "never-synced-model"); + + assert.equal( + selectedConnectionId(selected), + connectionId, + "an unsynced provider must not be filtered to zero candidates" + ); +}); diff --git a/tests/unit/chatcore-translation-paths.test.ts b/tests/unit/chatcore-translation-paths.test.ts index 08cc564cc4e..6ecb0487889 100644 --- a/tests/unit/chatcore-translation-paths.test.ts +++ b/tests/unit/chatcore-translation-paths.test.ts @@ -369,7 +369,7 @@ async function invokeChatCore({ onCredentialsRefreshed = null, onRequestSuccess = null, sessionAffinityKey = null, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", managedLease = null, cachedSettings = null, }: any = {}) { @@ -631,7 +631,7 @@ test("chatCore translates a streaming Responses upstream for a Chat client", asy assert.match(streamed, /"content":"ok"/); assert.match(streamed, /data: \[DONE\]/); }); -test("chatCore rejects opaque reasoning for unknown Responses targets unless explicitly enabled", async () => { +test("chatCore drops opaque reasoning for plaintext Responses targets by default (#10959)", async () => { const reasoningItems = [ { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }, { type: "reasoning", encrypted_content: "" }, @@ -640,7 +640,7 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp { id: "fc_call", type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, ]; - const rejected = await invokeChatCore({ + const dropped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/responses", @@ -656,9 +656,12 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp responseFormat: "openai-responses", }); - assert.equal(rejected.result.success, false); - assert.equal(rejected.result.status, 400); - assert.equal(rejected.calls.length, 0); + assert.equal(dropped.result.success, true); + assert.equal(dropped.calls.length, 1); + assert.deepEqual( + dropped.call.body.input.filter((item) => item.type === "reasoning"), + [{ type: "reasoning", summary: [{ text: "not self-contained" }] }] + ); const enabled = await invokeChatCore({ provider: "openai-compatible-sp-openai", @@ -682,7 +685,7 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp assert.deepEqual( input.filter((item) => item.type === "reasoning"), [ - { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }, + { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }, // summary defaulted by #11110 { type: "reasoning", summary: [{ text: "not self-contained" }] }, ] ); @@ -693,9 +696,9 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp assert.equal(input.find((item) => item.type === "function_call")?.id, undefined); }); -test("chatCore applies Chat reasoning compatibility before stream mode diverges", async () => { +test("chatCore drops incompatible Chat reasoning before stream mode diverges (#10959)", async () => { for (const stream of [false, true]) { - const rejected = await invokeChatCore({ + const dropped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/chat/completions", @@ -728,14 +731,14 @@ test("chatCore applies Chat reasoning compatibility before stream mode diverges" }, }); - assert.equal(rejected.result.success, false, `stream=${stream}`); - assert.equal(rejected.result.status, 400, `stream=${stream}`); - assert.equal(rejected.calls.length, 0, `stream=${stream}`); + assert.equal(dropped.result.success, true, `stream=${stream}`); + assert.equal(dropped.calls.length, 1, `stream=${stream}`); + assert.equal(dropped.call.body.messages[0].reasoning_details, undefined, `stream=${stream}`); } }); -test("chatCore can drop incompatible reasoning for an opted-in Combo attempt", async () => { - const dropped = await invokeChatCore({ +test("chatCore preserves Combo skip behavior for incompatible reasoning", async () => { + const skipped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/responses", @@ -757,15 +760,12 @@ test("chatCore can drop incompatible reasoning for an opted-in Combo attempt", a }, responseFormat: "openai-responses", isCombo: true, - reasoningTransportFallback: "drop", + reasoningTransportFallback: "skip", }); - assert.equal(dropped.result.success, true); - assert.equal(dropped.calls.length, 1); - assert.equal( - dropped.call.body.input.some((item) => item.type === "reasoning"), - false - ); + assert.equal(skipped.result.success, false); + assert.equal(skipped.result.status, 400); + assert.equal(skipped.calls.length, 0); }); test("chatCore carries Chat reasoning_content into official DeepSeek Responses input", async () => { @@ -800,6 +800,7 @@ test("chatCore carries Chat reasoning_content into official DeepSeek Responses i assert.deepEqual(call.body.input.slice(0, 3), [ { type: "reasoning", + summary: [], // defaulted on freshly-built reasoning items (#11129) content: [{ type: "reasoning_text", text: "Inspect before calling the tool" }], }, { @@ -866,7 +867,7 @@ test("chatCore replays nonstream DeepSeek Responses reasoning across a Chat tool assert.equal(second.result.success, true); assert.deepEqual( second.call.body.input.find((item) => item.type === "reasoning"), - { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }] } + { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], summary: [] } // summary defaulted by #11129 ); }); @@ -930,7 +931,7 @@ test("chatCore replays streamed DeepSeek Responses reasoning across a Chat tool assert.equal(second.result.success, true); assert.deepEqual( second.call.body.input.find((item) => item.type === "reasoning"), - { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }] } + { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], summary: [] } // summary defaulted by #11129 ); }); @@ -1132,7 +1133,7 @@ test("chatCore automatically preserves provider-generated opaque reasoning for C assert.equal(result.success, true); assert.deepEqual( call.body.input.filter((item) => item.type === "reasoning"), - [{ id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }] + [{ id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }] // summary defaulted by #11110 ); assert.equal( call.body.input.some((item) => item.type === "item_reference"), diff --git a/tests/unit/cli-combo-create-models-10954.test.ts b/tests/unit/cli-combo-create-models-10954.test.ts index dc53459d6ed..52fabf8c169 100644 --- a/tests/unit/cli-combo-create-models-10954.test.ts +++ b/tests/unit/cli-combo-create-models-10954.test.ts @@ -90,7 +90,15 @@ test("combo create — parses --models without throwing (Commander option regist }); await prog.parseAsync( - ["node", "x", "combo", "create", "my-combo", "--models", "openai/gpt-4o,anthropic/claude-3-opus"], + [ + "node", + "x", + "combo", + "create", + "my-combo", + "--models", + "openai/gpt-4o,anthropic/claude-3-opus", + ], { from: "node" } ); @@ -222,3 +230,26 @@ test("combo create (HTTP) — POST /api/combos body carries the parsed models", else process.env.DATA_DIR = ORIGINAL_DATA_DIR; } }); + +// Regression for the follow-up of #11011: with --models available, creating +// an empty combo is no longer a legitimate path on either transport. +test("combo create without any model is refused before reaching a transport", async () => { + await withComboEnv(async () => { + const errors: string[] = []; + const originalError = console.error; + console.error = (msg?: unknown) => { + errors.push(String(msg)); + }; + try { + const mod = await import("../../bin/cli/commands/combo.mjs"); + const rc = await mod.runComboCreateCommand("guard-test"); + assert.equal(rc, 1); + } finally { + console.error = originalError; + } + assert.ok( + errors.some((m) => m.includes("--models")), + `stderr should name --models, got: ${errors.join(" | ")}` + ); + }); +}); diff --git a/tests/unit/cli-helper/config-generator.test.ts b/tests/unit/cli-helper/config-generator.test.ts index 1b2e85732fe..cfe3f398096 100644 --- a/tests/unit/cli-helper/config-generator.test.ts +++ b/tests/unit/cli-helper/config-generator.test.ts @@ -414,7 +414,7 @@ describe("config-generator", () => { } }); - it("does NOT fabricate a default context when the catalog has no entry", async () => { + it("uses the required 128K context fallback when the catalog has no entry", async () => { const stub = stubFetchOnce(makeCatalogResponse(SAMPLE_CATALOG)); try { const { generateOpencodeConfig } = @@ -424,15 +424,14 @@ describe("config-generator", () => { apiKey: "sk-test", }); const cfg = JSON.parse(out); - // NO_CTX_COMBO has no context_length in the catalog — generator - // must NOT default to 128K (or any other value). The entry is - // emitted without limit.context so OpenCode's own heuristic - // applies and the user can fix the upstream. + // NO_CTX_COMBO has no context_length in the catalog. OpenCode v1 + // requires a complete limit object, so the compatibility fallback + // must be explicit rather than leaving the config invalid. const noCtx = cfg.provider.omniroute.models["NO_CTX_COMBO"]; assert.strictEqual( noCtx.limit?.context, - undefined, - `NO_CTX_COMBO should not have a fabricated limit.context (got ${noCtx.limit?.context})` + 128_000, + `NO_CTX_COMBO should use the 128K fallback (got ${noCtx.limit?.context})` ); } finally { stub.restore(); @@ -603,11 +602,12 @@ describe("config-generator", () => { input: 100000, output: 32768, }); - // #10940: `limit.output` is REQUIRED by OpenCode's v1 provider schema, - // so even a model with zero catalog metadata still gets a `limit` - // block carrying the fallback output value; `context`/`input` stay - // omitted since neither the catalog nor the user knows them. - assert.deepStrictEqual(models["no-metadata"].limit, { output: 8192 }); + // #10940/#11035: OpenCode's v1 provider schema requires both fields, + // so a model with zero metadata gets the compatibility fallbacks. + assert.deepStrictEqual(models["no-metadata"].limit, { + context: 128_000, + output: 8192, + }); for (const model of Object.values(models) as Array<{ limit?: { output?: number } }>) { assert.ok( diff --git a/tests/unit/codex-drop-nonstandard-events.test.ts b/tests/unit/codex-drop-nonstandard-events.test.ts index d2aff548149..dec24970ca5 100644 --- a/tests/unit/codex-drop-nonstandard-events.test.ts +++ b/tests/unit/codex-drop-nonstandard-events.test.ts @@ -17,6 +17,19 @@ function sseResponse(body: string): Response { }); } +function chunkedSseResponse(chunks: string[]): Response { + const encoder = new TextEncoder(); + return new Response( + new ReadableStream({ + start(controller) { + for (const chunk of chunks) controller.enqueue(encoder.encode(chunk)); + controller.close(); + }, + }), + { status: 200, headers: { "content-type": "text/event-stream" } } + ); +} + async function readAll(res: Response): Promise { return await res.text(); } @@ -61,10 +74,10 @@ describe("codexDropNonstandardEvents (#11014)", () => { describe("filterNonstandardCodexSse (#4715)", () => { it("drops codex.* event blocks but keeps standard response.* events", async () => { const stream = - "event: response.created\ndata: {\"type\":\"response.created\"}\n\n" + + 'event: response.created\ndata: {"type":"response.created"}\n\n' + "event: codex.rate_limits\n\n" + - "event: response.output_text.delta\ndata: {\"delta\":\"hi\"}\n\n" + - "event: response.completed\ndata: {\"type\":\"response.completed\"}\n\n"; + 'event: response.output_text.delta\ndata: {"delta":"hi"}\n\n' + + 'event: response.completed\ndata: {"type":"response.completed"}\n\n'; const out = await readAll(filterNonstandardCodexSse(sseResponse(stream))); assert.ok(!out.includes("codex.rate_limits"), "codex.* frame must be stripped"); assert.ok(out.includes("response.created"), "standard events preserved"); @@ -72,18 +85,31 @@ describe("filterNonstandardCodexSse (#4715)", () => { assert.ok(out.includes("response.completed"), "terminal event preserved"); }); + it("filters CRLF-framed events split across transport chunks", async () => { + const response = chunkedSseResponse([ + 'event: response.created\r\ndata: {"type":"response.created"}\r\n\r', + "\nevent: codex.rate_limits\r\n\r\n", + 'event: response.completed\r\ndata: {"type":"response.completed"}\r\n\r\n', + ]); + + const out = await readAll(filterNonstandardCodexSse(response)); + + assert.ok(!out.includes("codex.rate_limits"), "codex.* frame must be stripped"); + assert.ok(out.includes("response.created"), "standard events preserved"); + assert.ok(out.includes("response.completed"), "terminal event preserved"); + }); + it("passes through non-SSE responses untouched", async () => { - const json = new Response("{\"ok\":true}", { + const json = new Response('{"ok":true}', { status: 200, headers: { "content-type": "application/json" }, }); const out = filterNonstandardCodexSse(json); - assert.equal(await out.text(), "{\"ok\":true}"); + assert.equal(await out.text(), '{"ok":true}'); }); it("drops a trailing codex.* block with no double-newline terminator (flush path)", async () => { - const stream = - "event: response.created\ndata: {}\n\n" + "event: codex.token_count\ndata: {}"; + const stream = "event: response.created\ndata: {}\n\n" + "event: codex.token_count\ndata: {}"; const out = await readAll(filterNonstandardCodexSse(sseResponse(stream))); assert.ok(out.includes("response.created")); assert.ok(!out.includes("codex.token_count")); diff --git a/tests/unit/combo-bracket-names.test.ts b/tests/unit/combo-bracket-names.test.ts index 32844173cc1..0f4d8af0de9 100644 --- a/tests/unit/combo-bracket-names.test.ts +++ b/tests/unit/combo-bracket-names.test.ts @@ -30,6 +30,7 @@ test.after(() => { test("combo schemas accept names with spaces and square brackets", () => { const createResult = schemas.createComboSchema.safeParse({ name: "Claude [1m]", + models: ["anthropic/claude-3-opus"], }); const updateResult = schemas.updateComboSchema.safeParse({ name: "Claude [1m]", diff --git a/tests/unit/combo-context-length.test.ts b/tests/unit/combo-context-length.test.ts index f47539d64ad..c416b4d8be3 100644 --- a/tests/unit/combo-context-length.test.ts +++ b/tests/unit/combo-context-length.test.ts @@ -46,6 +46,7 @@ test.after(async () => { test("createComboSchema accepts valid context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 128000, }); assert.equal(result.success, true); @@ -54,6 +55,7 @@ test("createComboSchema accepts valid context_length", () => { test("createComboSchema rejects context_length below minimum (1000)", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 999, }); assert.equal(result.success, false); @@ -62,6 +64,7 @@ test("createComboSchema rejects context_length below minimum (1000)", () => { test("createComboSchema rejects context_length above maximum (2000000)", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 2000001, }); assert.equal(result.success, false); @@ -70,12 +73,14 @@ test("createComboSchema rejects context_length above maximum (2000000)", () => { test("createComboSchema accepts context_length at exact boundaries", () => { const min = schemas.createComboSchema.safeParse({ name: "MinCombo", + models: ["openai/gpt-4o-mini"], context_length: 1000, }); assert.equal(min.success, true); const max = schemas.createComboSchema.safeParse({ name: "MaxCombo", + models: ["openai/gpt-4o-mini"], context_length: 2000000, }); assert.equal(max.success, true); @@ -84,6 +89,7 @@ test("createComboSchema accepts context_length at exact boundaries", () => { test("createComboSchema rejects non-integer context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 128000.5, }); assert.equal(result.success, false); @@ -92,6 +98,7 @@ test("createComboSchema rejects non-integer context_length", () => { test("createComboSchema accepts omitted context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], }); assert.equal(result.success, true); }); diff --git a/tests/unit/combo-empty-models.test.ts b/tests/unit/combo-empty-models.test.ts index 1486628ec53..95f32813a34 100644 --- a/tests/unit/combo-empty-models.test.ts +++ b/tests/unit/combo-empty-models.test.ts @@ -24,9 +24,9 @@ test("an update cannot remove every model from a combo", () => { assert.equal(updateComboSchema.safeParse({ name: "renamed" }).success, true); }); -test("creating a combo with no model stays allowed — the CLI does it on purpose", () => { - assert.equal(createComboSchema.safeParse({ name: "drafted", models: [] }).success, true); - assert.equal(createComboSchema.safeParse({ name: "drafted" }).success, true); +test("creating a combo without a model is refused at the boundary", () => { + assert.equal(createComboSchema.safeParse({ name: "drafted", models: [] }).success, false); + assert.equal(createComboSchema.safeParse({ name: "drafted" }).success, false); }); test("the copilot createCombo tool stores targets where the router looks for them", async () => { diff --git a/tests/unit/combo-quota-exhaustion-only-fallback.test.ts b/tests/unit/combo-quota-exhaustion-only-fallback.test.ts index 2a90d7e3fbd..cbb4f70c11f 100644 --- a/tests/unit/combo-quota-exhaustion-only-fallback.test.ts +++ b/tests/unit/combo-quota-exhaustion-only-fallback.test.ts @@ -110,7 +110,7 @@ async function run( } test("quota classifier rejects terminal-looking evidence on ineligible statuses", async () => { - for (const status of [400, 401, 403, 404, 408, 409, 422, 500, 502, 503, 504]) { + for (const status of [400, 401, 404, 408, 409, 422, 500, 502, 503, 504]) { for (const terminal of ["insufficient_quota", "quota_exhausted", "credits_exhausted"]) { assert.equal( await isQuotaExhaustionResponse( diff --git a/tests/unit/combo-runtime-unit-concurrency.test.ts b/tests/unit/combo-runtime-unit-concurrency.test.ts index 36a14a59ce7..8ccebe4603e 100644 --- a/tests/unit/combo-runtime-unit-concurrency.test.ts +++ b/tests/unit/combo-runtime-unit-concurrency.test.ts @@ -28,8 +28,8 @@ const databases = db.pragma("database_list") as Array<{ file?: string; name?: st const activeDbPath = databases.find((database) => database.name === "main")?.file; assert.ok(activeDbPath, "test requires a file-backed main SQLite database"); assert.equal( - path.dirname(path.resolve(activeDbPath)), - path.resolve(TEST_DATA_DIR), + fs.realpathSync(path.dirname(path.resolve(activeDbPath))), + fs.realpathSync(path.resolve(TEST_DATA_DIR)), `active test database must be under TEST_DATA_DIR before inserts: ${activeDbPath}` ); diff --git a/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts b/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts index 0f4abd534e6..a646744871c 100644 --- a/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts +++ b/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts @@ -15,6 +15,7 @@ import { describe, it, before } from "node:test"; import assert from "node:assert/strict"; import { ccrEngine, + getCcrStoreStats, resetCcrStore, retrieveBlock, } from "../../../open-sse/services/compression/engines/ccr/index.ts"; @@ -54,39 +55,117 @@ describe("issue #7746 — CCR must not reduce the sole user prompt to a bare, un }); it("prompt fixture is realistically sized (>= default 600-char minChars)", () => { - assert.ok(REPORTER_PROMPT.length >= 600, `fixture must be >= 600 chars, got ${REPORTER_PROMPT.length}`); + assert.ok( + REPORTER_PROMPT.length >= 600, + `fixture must be >= 600 chars, got ${REPORTER_PROMPT.length}` + ); }); - it("does not leave the model with only the bare CCR marker when no retrieve tool is available", () => { + it("non-MCP caller: CCR skips entirely — the sole user prompt passes through verbatim", () => { resetCcrStore(); const body = makeOpenCodeStyleRequestBody(); const result = ccrEngine.apply(body as Record, { stepConfig: {} }); - assert.equal(result.compressed, true, "CCR compressed the sole user message (reproducing the report)"); - - const messages = result.body.messages as Array<{ role: string; content: string }>; - const compressedContent = messages[0].content; - const isBareMarkerOnly = /^\[CCR retrieve hash=[0-9a-f]{24} chars=\d+\]$/.test(compressedContent); - + // #7746 follow-up (forge review outage, 2026-08-22): the preamble guard was + // not enough — a non-MCP caller received "[CCR retrieve hash=...] markers" + // it had no tool to resolve (upstream saw 112 of ~3.6K tokens). The engine + // now refuses to replace content at all when tools[] lacks + // omniroute_ccr_retrieve: compressed=false, message content untouched. assert.equal( - isBareMarkerOnly, + result.compressed, false, - "BUG #7746: CCR replaced the ENTIRE sole user message with nothing but the bare " + - `[CCR retrieve hash=...] marker, permanently losing the original prompt for any ` + - `non-MCP caller that cannot resolve the marker. Got: ${JSON.stringify(compressedContent)}` + "CCR must not compress for a caller without the retrieve tool" ); + assert.equal(result.stats, null, "no stats when the engine is skipped"); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].role, "user", "message role must stay user"); + assert.equal( + messages[0].content, + REPORTER_PROMPT, + "sole user prompt must pass through verbatim" + ); + assert.equal(messages.length, 1, "no protocol instruction may be injected for non-MCP callers"); + // Guard regression check: if callerSupportsCcrRetrieve ever returned true + // here, the store would silently accumulate blocks no non-MCP caller can + // retrieve. After a skip the store must hold nothing for this principal. + assert.equal(getCcrStoreStats().entries, 0, "store must stay empty after a non-MCP skip"); }); - it("the original prompt remains fully retrievable by hash even after the guard applies", () => { + // tools:[] and unrelated tools are distinct caller shapes that must all be + // treated as non-MCP: an empty array and a foreign tool list both mean the + // retrieve tool is unreachable. + for (const label of ["empty tools array", "unrelated tools"] as const) { + it(`non-MCP caller with ${label}: CCR skips entirely`, () => { + resetCcrStore(); + const tools = + label === "empty tools array" + ? [] + : [ + { type: "function", function: { name: "get_weather" } }, + { type: "function", function: { name: "web_search" } }, + ]; + const body = { ...makeOpenCodeStyleRequestBody(), tools }; + const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + + assert.equal(result.compressed, false, `${label} must not compress`); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].content, REPORTER_PROMPT, "prompt passes through verbatim"); + assert.equal(messages.length, 1, "no protocol instruction injected"); + }); + } + + // A malformed body (tools as a non-array, or entries of unexpected shape) + // must fail OPEN — no compression, never a throw into the request pipeline. + for (const malformed of [ + { tools: "not-an-array" }, + { tools: [null, 42, "x"] }, + { tools: [{}, { type: "function" }] }, + ]) { + it(`malformed tools payload (${JSON.stringify(malformed.tools)}): engine skips without throwing`, () => { + resetCcrStore(); + const body = { ...makeOpenCodeStyleRequestBody(), ...malformed }; + const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + + assert.equal(result.compressed, false, "malformed tools must fail open (skip)"); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].content, REPORTER_PROMPT, "prompt passes through verbatim"); + assert.equal( + getCcrStoreStats().entries, + 0, + "store must stay empty after a malformed-tools skip" + ); + }); + } + + it("MCP-capable caller (tools[] advertises omniroute_ccr_retrieve): replacement still runs and stays retrievable", () => { resetCcrStore(); - const body = makeOpenCodeStyleRequestBody(); + const body = { + ...makeOpenCodeStyleRequestBody(), + tools: [{ type: "function", function: { name: "omniroute_ccr_retrieve" } }], + }; const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + assert.equal(result.compressed, true, "CCR still compresses for MCP-capable callers"); const messages = result.body.messages as Array<{ role: string; content: string }>; - const compressedContent = messages[0].content; + // The protocol instruction is injected as a leading system message, so the + // compressed conversation is exactly: [instruction, original user message]. + assert.equal(messages.length, 2, "instruction + user message"); + assert.equal(messages[0].role, "system", "instruction is a leading system message"); + assert.ok( + typeof messages[0].content === "string" && messages[0].content.length > 0, + "instruction content must be non-empty" + ); + assert.ok( + messages[0].content.includes("omniroute_ccr_retrieve"), + "instruction must teach the retrieve tool contract" + ); + const compressedContent = messages[1].content; const match = compressedContent.match(/\[CCR retrieve hash=([0-9a-f]{24}) chars=\d+\]/); - assert.ok(match, "compressed content must still contain a resolvable CCR marker"); - const hash = match![1]; - assert.equal(retrieveBlock(hash), REPORTER_PROMPT, "original prompt must be stored verbatim and retrievable"); + assert.ok(match, "compressed content must contain a resolvable CCR marker"); + assert.equal( + retrieveBlock(match![1]), + REPORTER_PROMPT, + "original prompt must be stored verbatim and retrievable" + ); }); }); diff --git a/tests/unit/compression/ccr-retrieval-ramp.test.ts b/tests/unit/compression/ccr-retrieval-ramp.test.ts index 2b475e69bfb..6e87f7b7877 100644 --- a/tests/unit/compression/ccr-retrieval-ramp.test.ts +++ b/tests/unit/compression/ccr-retrieval-ramp.test.ts @@ -78,7 +78,13 @@ describe("ccrEngine.apply — retrieval-aware compression (H8)", () => { const block = (len: number) => "x".repeat(len); const run = (content: string, retrievalRampFactor = 2) => ccrEngine.apply( - { messages: [{ role: "user", content }] }, + // The retrieve tool is advertised — this suite exercises the compression + // path itself (H8 ramp); without the tool declaration the #7746 guard + // skips the engine entirely. + { + messages: [{ role: "user", content }], + tools: [{ type: "function", function: { name: "omniroute_ccr_retrieve" } }], + }, { stepConfig: { minChars: BASE, retrievalRampFactor }, principalId: P } ); diff --git a/tests/unit/context-manager-purify-system-first.test.ts b/tests/unit/context-manager-purify-system-first.test.ts new file mode 100644 index 00000000000..51c658231d0 --- /dev/null +++ b/tests/unit/context-manager-purify-system-first.test.ts @@ -0,0 +1,91 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { compressContext } from "../../open-sse/services/contextManager.ts"; + +/** + * Plan-A root fix for the 2026-08-22 tokenrouter 400s. purifyHistory() used to + * splice the `[Context compressed: …]` notice as a SECOND system-role message at + * index system.length; strict gateways (TokenRouter, xiaomi-mimo/mimo) reject any + * system message at index > 0 with HTTP 400 "System message must be at the + * beginning". The notice must now merge into the leading system/developer + * message — or prepend a single system message when none exists — so the output + * never contains a system role after index 0, for ANY provider. + */ + +function bigTurn(n: number) { + return { role: "user", content: `turn ${n}: ${"x".repeat(4_000)}` }; +} + +function run(body: Record) { + // ~30k tokens of history vs a small target forces Layer-3 purify_history. + return compressContext(body, { maxTokens: 5_000, reserveTokens: 0 }); +} + +function systemIndices(messages: Array<{ role: string }>) { + return messages.map((m, i) => (m.role === "system" ? i : -1)).filter((i) => i >= 0); +} + +test("purify_history merges dropped-notice into existing leading system message", () => { + const body = { + model: "any-model", + messages: [ + { role: "system", content: "You are a helpful assistant." }, + ...Array.from({ length: 12 }, (_, i) => bigTurn(i)), + ], + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual(systemIndices(messages as Array<{ role: string }>).slice(1), []); + const first = messages[0]; + assert.equal(first.role, "system"); + const text = String(first.content); + assert.match(text, /Context compressed: \d+ earlier messages removed/); + assert.match(text, /You are a helpful assistant\./); +}); + +test("purify_history prepends a single system notice when no system message exists", () => { + const body = { + model: "any-model", + messages: Array.from({ length: 12 }, (_, i) => bigTurn(i)), + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual(systemIndices(messages as Array<{ role: string }>), [0]); + assert.match(String(messages[0].content), /Context compressed: \d+ earlier messages removed/); +}); + +test("purify_history merges into leading developer message without adding a second one", () => { + const body = { + model: "any-model", + messages: [ + { role: "developer", content: "dev instructions" }, + ...Array.from({ length: 12 }, (_, i) => bigTurn(i)), + ], + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual( + messages.filter((m) => m.role === "developer").length, + 1, + "exactly one developer message" + ); + assert.match(String(messages[0].content), /Context compressed: \d+ earlier messages removed/); + assert.match(String(messages[0].content), /dev instructions/); +}); + +test("no compression means no notice and untouched history", () => { + const body = { + model: "any-model", + messages: [ + { role: "system", content: "sys" }, + { role: "user", content: "hi" }, + ], + }; + const result = run(body); + assert.equal(result.compressed, false); + const messages = (result.body as { messages: unknown[] }).messages; + assert.equal(messages.length, 2); +}); diff --git a/tests/unit/cursor-image-input.test.ts b/tests/unit/cursor-image-input.test.ts index 16bffcb6b30..abc14d7d983 100644 --- a/tests/unit/cursor-image-input.test.ts +++ b/tests/unit/cursor-image-input.test.ts @@ -1,4 +1,4 @@ -import test from "node:test"; +import test, { type TestContext } from "node:test"; import assert from "node:assert/strict"; import crypto from "node:crypto"; import dns from "node:dns"; @@ -473,7 +473,37 @@ test("resolveCursorImages soft-caps a large PNG under the wire budget", async () // ─── Executor-level error body (response path, hard rule #12) ─────────────── -test("executor returns a sanitized 400 for an oversized image", async () => { +// #10804 moved agent-endpoint discovery (a live api2.cursor.sh call) ahead of +// request building inside CursorExecutor.execute. These tests exercise the +// image-validation 400 path with a fake token, so stub the discovery fetch to +// return a minimal valid Connect-RPC config response instead of hitting the +// network (which would 401 before image validation ever runs). +function mockCursorServerConfig(t: TestContext): void { + t.mock.method(globalThis, "fetch", async (input, init) => { + const url = String(input); + if (!url.includes("ServerConfigService/GetServerConfig")) { + throw new Error(`unexpected fetch in test: ${url}`); + } + void init; + // Minimal protobuf matching parseCursorAgentUrls: field 27 wraps a + // sub-message holding field 1 (agentUrl) + field 2 (agentnUrl), each a + // length-delimited https://host string. validateCursorAgentUrl only + // accepts *.api5.cursor.sh hosts, so use those. + const str = (field: number, host: string): Buffer => { + const value = Buffer.from(`https://${host}`); + return Buffer.concat([Buffer.from([(field << 3) | 0x02, value.length]), value]); + }; + const inner = Buffer.concat([str(1, "us.api5.cursor.sh"), str(2, "eu.api5.cursor.sh")]); + // Field-27 tag (218) needs proper varint encoding (2 bytes). + const tag = ((27 << 3) | 0x02) as number; + const header = Buffer.from([(tag & 0x7f) | 0x80, tag >>> 7, inner.length]); + const body = Buffer.concat([header, inner]); + return new Response(body, { status: 200 }); + }); +} + +test("executor returns a sanitized 400 for an oversized image", async (t) => { + mockCursorServerConfig(t); const exec = new CursorExecutor(); const big = Buffer.alloc(MAX_CURSOR_IMAGE_DECODE_BYTES + 16).toString("base64"); const result = await exec.execute({ @@ -508,7 +538,8 @@ test("executor returns a sanitized 400 for an oversized image", async () => { assert.ok(!/\/(root|home|usr)\//.test(body.error.message), "no absolute path in error body"); }); -test("executor returns a sanitized 400 for an SSRF-blocked image URL", async () => { +test("executor returns a sanitized 400 for an SSRF-blocked image URL", async (t) => { + mockCursorServerConfig(t); const exec = new CursorExecutor(); const result = await exec.execute({ model: "gpt-5.2", diff --git a/tests/unit/dashboard-ux-operability.test.ts b/tests/unit/dashboard-ux-operability.test.ts new file mode 100644 index 00000000000..cd3e3b533ad --- /dev/null +++ b/tests/unit/dashboard-ux-operability.test.ts @@ -0,0 +1,9 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { formatQuotaLabel } from "../../src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx"; + +test("formatQuotaLabel formats custom quota keys with proper title-casing", () => { + assert.equal(formatQuotaLabel("session"), "Session"); + assert.equal(formatQuotaLabel("weekly"), "Weekly"); + assert.equal(formatQuotaLabel("custom_quota_limit"), "Custom Quota Limit"); +}); diff --git a/tests/unit/discontinued-providers-2026.test.ts b/tests/unit/discontinued-providers-2026.test.ts index 1979a7cd9aa..88f963ef911 100644 --- a/tests/unit/discontinued-providers-2026.test.ts +++ b/tests/unit/discontinued-providers-2026.test.ts @@ -6,6 +6,9 @@ import assert from "node:assert"; // free tier that does not exist. The budget catalog already dropped them. The 2026-06-18 batch // (gitlawb, gitlawb-gmi, aimlapi, yi) was each re-verified against the official source before flipping // (aimlapi docs: "The Free Tier is currently paused"; gitlawb GitHub issue #1345: MiMo revoked). +// 2026-08-22 (#10071): the five g4f.space sub-providers lost their anonymous tier to a proof-of-work +// credit wall (keyless POST -> HTTP 402 insufficient_credits). They remain usable with a g4f.dev +// member key, so only hasFree/freeNote/authHint changed - registry wiring is untouched. describe("2026 discontinued free tiers — providers.ts hasFree reconciliation", () => { it("APIKEY_PROVIDERS dead tiers no longer advertise a free tier", async () => { const { APIKEY_PROVIDERS } = await import("../../src/shared/constants/providers.ts"); @@ -27,6 +30,42 @@ describe("2026 discontinued free tiers — providers.ts hasFree reconciliation", } }); + it("g4f.space sub-providers no longer advertise an anonymous free tier", async () => { + const { APIKEY_PROVIDERS } = await import("../../src/shared/constants/providers.ts"); + // 2026-08-22 live re-verification: every g4f.space sub-path still lists models keylessly, but a + // keyless POST /v1/chat/completions returns HTTP 402 {"type":"insufficient_credits"} pointing at + // a proof-of-work "cake" wall (g4f.dev/chat) or a member key (g4f.dev/members.html). The gateway + // is NOT dead - it works with a g4f.dev member key - so the registry entries and their + // authType:"optional" are deliberately untouched; only the free-tier advertisement is corrected. + for (const id of ["g4f-groq", "g4f-gemini", "g4f-pollinations", "g4f-ollama", "g4f-nvidia"]) { + const p = ( + APIKEY_PROVIDERS as Record< + string, + { hasFree?: boolean; freeNote?: string; authHint?: string } + > + )[id]; + assert.ok( + p, + `${id} should still exist in APIKEY_PROVIDERS (gateway still usable with a member key)` + ); + assert.strictEqual( + p.hasFree, + false, + `${id} should have hasFree:false (anonymous tier walled behind proof-of-work credits in 2026)` + ); + assert.match( + p.freeNote ?? "", + /proof-of-work/i, + `${id} freeNote should explain the proof-of-work credit wall` + ); + assert.match( + p.authHint ?? "", + /member key/i, + `${id} authHint should state that a g4f.dev member key is required` + ); + } + }); + it("phind is fully removed (service shut down 2026-01) from both catalogs", async () => { const { APIKEY_PROVIDERS, WEB_COOKIE_PROVIDERS } = await import("../../src/shared/constants/providers.ts"); diff --git a/tests/unit/empty-stream-no-content-8649.test.ts b/tests/unit/empty-stream-no-content-8649.test.ts index e46157fbe17..918ae7c4c06 100644 --- a/tests/unit/empty-stream-no-content-8649.test.ts +++ b/tests/unit/empty-stream-no-content-8649.test.ts @@ -252,3 +252,68 @@ test("#8649 buildStreamErrorChunks-shaped error must not be rewritten as empty c assert.match(text, /AI Model Not Found/); assert.doesNotMatch(text, /Provider returned empty content/); }); + +test("#8649 a Responses compaction-only stream is real output, not empty content", async () => { + // Codex remote compaction V2: POST /v1/responses with a compaction_trigger + // input item completes with output = [{type:"compaction", encrypted_content}] + // and no assistant text. The watcher's content keys do not include + // encrypted_content, so the healthy stream was followed by a synthetic + // response.failed ("Provider returned empty content") — which strict + // Responses clients reject even after response.completed. + const text = await runClientStream( + [ + `data: {"type":"response.in_progress"}\n\n`, + `event: response.created\ndata: ${JSON.stringify({ + type: "response.created", + response: { id: "resp_cmp", status: "in_progress", output: [] }, + })}\n\n`, + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp_cmp", + status: "completed", + output: [ + { id: "cmp_1", type: "compaction", encrypted_content: "gAAAAABencryptedpayload" }, + ], + }, + })}\n\n`, + ], + FORMATS.OPENAI_RESPONSES + ); + + assert.match(text, /"type":"compaction"/); + assert.doesNotMatch( + text, + /Provider returned empty content|response\.failed/, + "a completed compaction response must not be followed by a synthetic failure frame" + ); +}); + +test("#8649 an encrypted-reasoning-only stream is still empty content", async () => { + // Inverse of the compaction carve-out: an encrypted reasoning item is not + // user-visible output. A turn that produces only a reasoning trace and no + // message/tool call is the fake-success shape this guard exists to catch. + const text = await runClientStream( + [ + `event: response.created\ndata: ${JSON.stringify({ + type: "response.created", + response: { id: "resp_r", status: "in_progress", output: [] }, + })}\n\n`, + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp_r", + status: "completed", + output: [{ id: "rs_1", type: "reasoning", encrypted_content: "gAAAAABencryptedtrace" }], + }, + })}\n\n`, + ], + FORMATS.OPENAI_RESPONSES + ); + + assert.match( + text, + /response\.failed|Provider returned empty content/, + "a reasoning-only turn must keep tripping the empty-content guard" + ); +}); diff --git a/tests/unit/flat-rate-cost-5552.test.ts b/tests/unit/flat-rate-cost-5552.test.ts index 2df378ee3e5..61ad9847b16 100644 --- a/tests/unit/flat-rate-cost-5552.test.ts +++ b/tests/unit/flat-rate-cost-5552.test.ts @@ -25,6 +25,7 @@ test("isFlatRateProvider: dedicated subscription / coding-plan providers are fla "glm-cn", "claude", "cc", + "opencode-go", ]) { assert.equal(isFlatRateProvider(id), true, `${id} should be flat-rate`); } @@ -83,6 +84,27 @@ test("computeCostFromPricing: opt-in only — flat-rate provider WITHOUT the fla assert.equal(computeCostFromPricing(PRICING, TOKENS, { provider: "chatgpt-web" }), 3); }); +test("#11149: opencode-go is a flat-rate subscription, not metered", () => { + // opencode-go (https://opencode.ai/go) is a $10/month flat subscription that + // resells GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x. Because it is + // an aggregator, every call was priced at the UNDERLYING model's metered rate, + // so the overstatement is large rather than marginal (a reported ~$13.35 for a + // month actually billed at $10 flat). It is api-key auth, so it is not covered + // by the dynamic WEB_COOKIE_PROVIDERS branch and needs the explicit id. + assert.equal(isFlatRateProvider("opencode-go"), true); + assert.equal( + computeCostFromPricing(PRICING, TOKENS, { provider: "opencode-go", flatRateAsZero: true }), + 0 + ); + // Still opt-in: without the flag the per-request estimate is unchanged. + assert.equal(computeCostFromPricing(PRICING, TOKENS, { provider: "opencode-go" }), 3); +}); + +test("#11149: sibling opencode ids keep their own billing semantics", () => { + // Only the Go subscription is flat-rate. The keyless `opencode` provider is a + // different id and must not be swept in by a prefix-style match. + assert.equal(isFlatRateProvider("opencode"), false); +}); test("computeCostFromPricing: metered provider with the flag still estimates", () => { assert.equal( computeCostFromPricing(PRICING, TOKENS, { provider: "openai", flatRateAsZero: true }), diff --git a/tests/unit/g4f-space-gateway-6650.test.ts b/tests/unit/g4f-space-gateway-6650.test.ts index 8f1aadc2bcd..93fd99bd679 100644 --- a/tests/unit/g4f-space-gateway-6650.test.ts +++ b/tests/unit/g4f-space-gateway-6650.test.ts @@ -17,7 +17,8 @@ * category on the dashboard * - allowed to skip API key validation (providerAllowsOptionalApiKey) * - has provider metadata (name/website/free-tier note) in the apikey - * gateway catalog + * gateway catalog (hasFree flipped false by #10071 — anonymous tier now + * requires proof-of-work credits; a g4f.dev member key is required) */ import test from "node:test"; import assert from "node:assert/strict"; @@ -96,7 +97,10 @@ for (const [id, subPath] of Object.entries(SUB_PATHS)) { assert.ok(meta, `${id} should have an APIKEY_PROVIDERS metadata entry`); assert.equal(meta.id, id); assert.equal(meta.website, "https://g4f.space"); - assert.equal(meta.hasFree, true); + // hasFree was true at #6650 time; the anonymous tier was walled behind proof-of-work + // credits in 2026 (#10071), so the flag is now false. Registry wiring above is unchanged: + // the provider still works with a g4f.dev member key, hence authType stays "optional". + assert.equal(meta.hasFree, false); assert.equal(typeof meta.freeNote, "string"); assert.ok((meta.freeNote as string).length > 0); }); diff --git a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts index d14c136e7b3..5d927de02a8 100644 --- a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts +++ b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts @@ -1,7 +1,7 @@ import test from "node:test"; import assert from "node:assert/strict"; -// GLM-5.3 support (released 2026-08-14, https://z.ai/blog/glm-5.3). +// GLM-5.3 support (released 2026-08-14, https://docs.z.ai/guides/llm/glm-5.3). // // Upstream ships ONE model id (`glm-5.3`) — effort is a request parameter // (`reasoning_effort`: low|high|max, default max) on the coding chat/completions @@ -12,14 +12,15 @@ import assert from "node:assert/strict"; // beta header), the 5.3 tiers use the documented `reasoning_effort` param on the // OpenAI coding transport. // -// Spec caveat: Z.ai has not yet published the default context window — 1M is -// mirrored from GLM-5.2 (same base model) per operator decision; correct when -// the official spec lands. +// Z.AI documents a 1M context window and 128K maximum output. -const { getRegistryEntry } = await import("../../open-sse/config/providerRegistry.ts"); +const { getRegistryEntry, REGISTRY } = await import("../../open-sse/config/providerRegistry.ts"); const { GlmExecutor } = await import("../../open-sse/executors/glm.ts"); const { MODEL_SPECS } = await import("../../src/shared/constants/modelSpecs.ts"); const { GLM_PRICING } = await import("../../src/shared/constants/pricing/shared-tiers.ts"); +const metadataRegistry = await import("../../src/lib/modelMetadataRegistry.ts"); +const { shouldExposeSyncedEffortVariants, SYNCED_EFFORT_SKIP_PROVIDERS } = + await import("../../open-sse/utils/syncedEffortVariants.ts"); const GLM_5_3_IDS = ["glm-5.3", "glm-5.3-high", "glm-5.3-low"] as const; @@ -38,6 +39,87 @@ function modelIds(provider: string): string[] { return (entry.models ?? []).map((m) => m.id); } +test("shared GLM providers keep their dedicated aliases instead of synthesizing another layer", () => { + for (const provider of ["glm", "glm-cn", "glmt"]) { + assert.ok(SYNCED_EFFORT_SKIP_PROVIDERS.has(provider), provider); + assert.equal( + shouldExposeSyncedEffortVariants({ + id: `${provider}/glm-5.3`, + owned_by: provider, + capabilities: { effort_tiers: ["low", "high", "max"] }, + }), + false, + provider + ); + } + assert.equal(SYNCED_EFFORT_SKIP_PROVIDERS.has("zcode"), false); +}); + +test("GLM family detection covers numeric, Z1, and bare provider model ids", () => { + for (const modelId of [ + "hf:zai-org/GLM-5.2", + "THUDM/GLM-Z1-32B-0414", + "THUDM/GLM-Z1-9B-0414", + "glm", + ]) { + assert.equal(metadataRegistry.isGlmFamilyModel(modelId), true, modelId); + } + assert.equal(metadataRegistry.isGlmFamilyModel("llama-3.3"), false); +}); + +test("catalog suppresses inferred tiers for every GLM registry entry without a provider contract", () => { + let audited = 0; + for (const [provider, entry] of Object.entries(REGISTRY)) { + for (const model of entry.models ?? []) { + if (!metadataRegistry.isGlmFamilyModel(model.id, model.name)) continue; + audited += 1; + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + if (capabilities.supportsThinking === true) { + assert.deepEqual( + capabilities.effort_tiers, + model.supportedThinkingEfforts ?? [], + `${provider}/${model.id}` + ); + } else { + assert.equal("effort_tiers" in capabilities, false, `${provider}/${model.id}`); + } + } + } + assert.ok(audited > 0); +}); + +test("catalog exposes only GLM effort tiers that each provider can route", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt", "zcode"]) { + for (const model of getRegistryEntry(provider)!.models ?? []) { + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + const expected = provider === "zcode" ? [] : (routedTiers.get(model.id) ?? []); + assert.equal(capabilities.supportsThinking, true, `${provider}/${model.id}`); + assert.deepEqual(capabilities.effort_tiers, expected, `${provider}/${model.id}`); + } + } +}); + for (const provider of ["glm", "glm-cn", "glmt"]) { test(`${provider} advertises the GLM-5.3 base model and effort tiers (GLM_SHARED_MODELS)`, () => { const ids = modelIds(provider); diff --git a/tests/unit/glm-executor.test.ts b/tests/unit/glm-executor.test.ts index 4c1a4942352..3d47b375c35 100644 --- a/tests/unit/glm-executor.test.ts +++ b/tests/unit/glm-executor.test.ts @@ -164,7 +164,12 @@ test("GlmExecutor separates OpenAI-compatible coding headers from Anthropic head const anthropicHeaders = executor.buildHeaders( { apiKey: "glm-key", - providerSpecificData: { baseUrl: "https://api.z.ai/api/anthropic/v1/messages" }, + providerSpecificData: { + baseUrl: "https://api.z.ai/api/anthropic/v1/messages", + // Same #10798 signature change — Anthropic transport via + // providerSpecificData (baseUrl is anthropic-shaped anyway). + primaryTransport: "anthropic", + }, }, true, null, @@ -191,6 +196,8 @@ test("GlmExecutor preserves extra API key rotation", () => { connectionId: "glm-rotation-test", providerSpecificData: { baseUrl: "https://api.z.ai/api/anthropic/v1/messages", + // #10798 signature change — Anthropic transport via providerSpecificData. + primaryTransport: "anthropic", extraApiKeys: ["extra-key"], }, }, @@ -426,10 +433,9 @@ test("GlmExecutor falls back internally to Anthropic transport and returns OpenA assert.equal(calls[0].url, "https://api.z.ai/api/coding/paas/v4/chat/completions"); assert.equal(calls[0].headers.Authorization, "Bearer glm-key"); assert.equal(calls[1].url, "https://api.z.ai/api/anthropic/v1/messages?beta=true"); - const fallbackKey = - calls[1].headers["x-api-key"] || - String(calls[1].headers.Authorization || "").replace(/^Bearer\s+/i, ""); - assert.equal(fallbackKey, "glm-key"); + assert.equal(calls[1].headers["x-api-key"], "glm-key"); + assert.equal(calls[1].headers.Authorization, undefined); + assert.equal(calls[1].headers["anthropic-version"], "2023-06-01"); assert.equal(calls[1].body.messages[0].role, "user"); assert.equal(calls[1].body._disableToolPrefix, undefined); assert.equal(result.targetFormat, "openai"); diff --git a/tests/unit/guide-settings-route.test.ts b/tests/unit/guide-settings-route.test.ts index 735240dc1e9..7516bd0be5c 100644 --- a/tests/unit/guide-settings-route.test.ts +++ b/tests/unit/guide-settings-route.test.ts @@ -198,8 +198,14 @@ test("guide-settings POST preserves existing OpenCode config fields while only u assert.equal(content.provider.omniroute.options.baseURL, "http://my-omni/v1"); assert.ok(content.provider.omniroute.options.apiKey.startsWith("sk-")); assert.deepEqual(content.provider.omniroute.models, { - "cx/gpt-5.6-sol": { name: "GPT-5.6 Sol" }, - "opencode-go/kimi-k2.6": { name: "Kimi K2.6" }, + "cx/gpt-5.6-sol": { + name: "GPT-5.6 Sol", + limit: { context: 128_000, output: 8192 }, + }, + "opencode-go/kimi-k2.6": { + name: "Kimi K2.6", + limit: { context: 128_000, output: 8192 }, + }, }); }); diff --git a/tests/unit/hard-session-lease-bypass-inventory.test.ts b/tests/unit/hard-session-lease-bypass-inventory.test.ts index bafa2971460..396c54015c7 100644 --- a/tests/unit/hard-session-lease-bypass-inventory.test.ts +++ b/tests/unit/hard-session-lease-bypass-inventory.test.ts @@ -86,7 +86,7 @@ const EXPECTED: Record> = { "src/app/api/providers/client/route.ts": 1, "src/app/api/providers/free-onboarding/route.ts": 2, "src/app/api/providers/import/route.ts": 1, - "src/app/api/providers/route.ts": 4, + "src/app/api/providers/route.ts": 2, "src/app/api/providers/test-batch/route.ts": 2, "src/app/api/rate-limits/route.ts": 1, "src/app/api/services/dario/admin/import-from-omniroute/route.ts": 2, diff --git a/tests/unit/lkgp-enabled-context-11181.test.ts b/tests/unit/lkgp-enabled-context-11181.test.ts new file mode 100644 index 00000000000..04950c2c2a5 --- /dev/null +++ b/tests/unit/lkgp-enabled-context-11181.test.ts @@ -0,0 +1,141 @@ +/** + * #11181 — the `lkgpEnabled` settings toggle must actually reach RoutingContext. + * + * `LKGPStrategyImpl.select()` guards with `context.lkgpEnabled === false` + * (open-sse/services/autoCombo/routerStrategy.ts), and the setting is persisted + * by the Routing settings tab (src/shared/validation/settingsSchemas.ts). But the + * RoutingContext literal built in resolveAutoStrategyOrder() never carried the + * field, so the guard could never fire in production. + * + * tests/unit/router-strategies.test.ts already covers the guard — but it hands + * the strategy a context it built itself, so it stays green whether or not the + * production construction site populates the field. These tests drive + * resolveAutoStrategyOrder() with a *persisted* setting instead, which is the + * only level at which the wiring is observable. + */ +import { test, after } from "node:test"; +import assert from "node:assert/strict"; + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-lkgp-11181-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const { resolveAutoStrategyOrder } = + await import("@omniroute/open-sse/services/combo/resolveAutoStrategy.ts"); +const settingsDb = await import("@/lib/db/settings.ts"); +const { resetDbInstance } = await import("@/lib/db/core.ts"); + +after(() => { + resetDbInstance(); +}); + +const target = (provider: string, modelStr: string): never => + ({ + kind: "model", + stepId: "s1", + executionKey: `${provider}>${modelStr}`, + modelStr, + provider, + providerId: null, + connectionId: null, + weight: 1, + label: null, + }) as never; + +const candidate = (provider: string, model: string, overrides: Record = {}) => ({ + kind: "model", + stepId: "s1", + executionKey: `${provider}>${model}`, + modelStr: model, + provider, + model, + quotaRemaining: 100, + quotaTotal: 100, + circuitBreakerState: "CLOSED", + costPer1MTokens: 1, + p95LatencyMs: 100, + latencyStdDev: 10, + errorRate: 0, + ...overrides, +}); + +// "cheap" wins under the rules scorer (cheapest + fastest + most stable); +// "pricey" only ever wins by being the persisted last-known-good provider. +const candidates = () => + [ + candidate("openai", "cheap-model", { + costPer1MTokens: 0.01, + p95LatencyMs: 10, + latencyStdDev: 1, + }), + candidate("anthropic", "pricey-model", { + costPer1MTokens: 50, + p95LatencyMs: 5000, + latencyStdDev: 900, + }), + ] as never; + +function capturingLog() { + const entries: string[] = []; + const push = (_tag: unknown, msg: unknown) => entries.push(String(msg)); + return { entries, info: push, warn: push, error: push, debug: push }; +} + +async function runWithSettings(comboName: string, settings: Record | null) { + // The LKGP pin resolveAutoStrategyOrder reads is getLKGP(combo.name, combo.id || combo.name). + await settingsDb.setLKGP(comboName, comboName, "anthropic"); + + const log = capturingLog(); + const result = await resolveAutoStrategyOrder({ + orderedTargets: [target("openai", "cheap-model"), target("anthropic", "pricey-model")], + body: { messages: [{ role: "user", content: "hi" }] }, + combo: { + id: comboName, + name: comboName, + autoConfig: { + routerStrategy: "lkgp", + candidatePool: ["openai", "anthropic"], + explorationRate: 0, + }, + }, + settings, + config: {}, + relayOptions: null, + resilienceSettings: { quotaPreflight: { enabled: false } }, + log, + buildAutoCandidates: (async () => candidates()) as never, + } as never); + + assert.ok("orderedTargets" in result, "expected an ordering result, not an earlyResponse"); + const selection = log.entries.find((entry) => entry.startsWith("Auto selection:")) ?? ""; + return { result, selection }; +} + +test("control — with lkgpEnabled unset the LKGP pin still wins (guard must not over-fire)", async () => { + const { result, selection } = await runWithSettings("lkgp-11181-default", null); + assert.match(selection, /LKGP: using last known good provider anthropic/); + if ("orderedTargets" in result) { + assert.equal(result.orderedTargets[0].provider, "anthropic"); + } +}); + +test("#11181 — a persisted lkgpEnabled:false makes the lkgp strategy delegate to rules", async () => { + const { result, selection } = await runWithSettings("lkgp-11181-disabled", { + lkgpEnabled: false, + }); + + // The whole point of the toggle: the LKGP pin must be ignored and the 6-factor + // rules scorer must pick the winner instead. + assert.doesNotMatch( + selection, + /LKGP: using last known good provider/, + `lkgpEnabled:false must disable LKGP selection, got: ${selection}` + ); + assert.match(selection, /RulesStrategy: score=/, `expected rules fallback, got: ${selection}`); + if ("orderedTargets" in result) { + assert.equal(result.orderedTargets[0].provider, "openai"); + } +}); diff --git a/tests/unit/logfare-registry.test.ts b/tests/unit/logfare-registry.test.ts new file mode 100644 index 00000000000..d5439cd1b23 --- /dev/null +++ b/tests/unit/logfare-registry.test.ts @@ -0,0 +1,63 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { logfareProvider } from "../../open-sse/config/providers/registry/logfare/index.ts"; + +const { APIKEY_PROVIDERS } = await import( + "../../src/shared/constants/providers.ts" +); +const { REGISTRY: providerRegistry } = + await import("../../open-sse/config/providerRegistry.ts"); +const { NAMED_OPENAI_STYLE_PROVIDERS, isNamedOpenAIStyleProvider } = + await import( + "../../src/app/api/providers/[id]/models/discovery/providerSets.ts" + ); + +const SPEC = { + id: "logfare", + alias: "logfare", + name: "Logfare", + website: "https://logfare.ai", + chatUrl: "https://logfare.ai/v1/chat/completions", + modelsUrl: "https://logfare.ai/v1/models", +}; + +test("logfareProvider registry entry has correct configuration", () => { + assert.equal(logfareProvider.id, "logfare"); + assert.equal(logfareProvider.alias, "logfare"); + assert.equal(logfareProvider.format, "openai"); + assert.equal(logfareProvider.executor, "default"); + assert.equal(logfareProvider.baseUrl, SPEC.chatUrl); + assert.equal(logfareProvider.modelsUrl, SPEC.modelsUrl); + assert.equal(logfareProvider.authType, "apikey"); + assert.equal(logfareProvider.authHeader, "bearer"); + // Catalog is discovered live from /v1/models; no hardcoded seed. + assert.equal(logfareProvider.passthroughModels, true); + assert.equal(logfareProvider.models.length, 0); +}); + +test("APIKEY_PROVIDERS.logfare is registered with the canonical identity", () => { + const entry = APIKEY_PROVIDERS[SPEC.id]; + assert.ok(entry, `APIKEY_PROVIDERS.${SPEC.id} must be defined`); + assert.equal(entry.id, SPEC.id); + assert.equal(entry.alias, SPEC.alias); + assert.equal(entry.name, SPEC.name); + assert.equal(entry.website, SPEC.website); + assert.equal(entry.hasFree, true); + assert.equal(typeof entry.freeNote, "string"); + assert.equal(typeof entry.apiHint, "string"); + assert.match(entry.color, /^#[0-9A-Fa-f]{6}$/); +}); + +test("providerRegistry exposes the OpenAI-compatible chat completions URL", () => { + assert.equal(providerRegistry[SPEC.id].baseUrl, SPEC.chatUrl); + assert.equal(providerRegistry[SPEC.id].modelsUrl, SPEC.modelsUrl); +}); + +test("logfare is classified as a named OpenAI-style provider (live-fetch path)", () => { + assert.ok( + NAMED_OPENAI_STYLE_PROVIDERS.has(SPEC.id), + "logfare must be in NAMED_OPENAI_STYLE_PROVIDERS for live /v1/models fetch" + ); + assert.equal(isNamedOpenAIStyleProvider(SPEC.id), true); +}); diff --git a/tests/unit/login-11143.test.ts b/tests/unit/login-11143.test.ts new file mode 100644 index 00000000000..f55f2edecf3 --- /dev/null +++ b/tests/unit/login-11143.test.ts @@ -0,0 +1,21 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import fs from "node:fs"; +import path from "node:path"; + +test("login page performs full window.location navigation after authentication to avoid cookie race", () => { + const loginPagePath = path.resolve(process.cwd(), "src/app/login/page.tsx"); + const content = fs.readFileSync(loginPagePath, "utf8"); + + // Ensure router.push("/dashboard") is replaced with window.location.href + assert.equal( + content.includes('router.push("/dashboard")'), + false, + "LoginPage should not use router.push('/dashboard') after login" + ); + assert.equal( + content.includes('window.location.href = "/dashboard"'), + true, + "LoginPage must perform full window.location navigation after login" + ); +}); diff --git a/tests/unit/memory-system-first-6135.test.ts b/tests/unit/memory-system-first-6135.test.ts index 7104007ae64..339f83792cf 100644 --- a/tests/unit/memory-system-first-6135.test.ts +++ b/tests/unit/memory-system-first-6135.test.ts @@ -49,6 +49,11 @@ describe("injectMemory system-must-be-first (#6135)", () => { it("flags xiaomi-mimo (and alias mimo) as system-must-be-first", () => { assert.equal(systemMessageMustBeFirst("xiaomi-mimo"), true); assert.equal(systemMessageMustBeFirst("mimo"), true); + // tokenrouter: confirmed live 2026-08-22 — mid-array system message + // (e.g. the purifyHistory compression notice) -> HTTP 400 + // "System message must be at the beginning". + assert.equal(systemMessageMustBeFirst("tokenrouter"), true); + assert.equal(systemMessageMustBeFirst("TokenRouter"), true); // case-insensitive // default: unlisted providers keep current (non-first-constrained) behavior assert.equal(systemMessageMustBeFirst("anthropic"), false); assert.equal(systemMessageMustBeFirst(null), false); diff --git a/tests/unit/mitm-cert-install-mode-9442.test.ts b/tests/unit/mitm-cert-install-mode-9442.test.ts index 979dee7d4c8..4128719c5cd 100644 --- a/tests/unit/mitm-cert-install-mode-9442.test.ts +++ b/tests/unit/mitm-cert-install-mode-9442.test.ts @@ -165,11 +165,14 @@ test("filesystem proof: cp under umask 0077 creates mode 0600 (why the fix is ne const oldUmask = process.umask(0o077); try { - // Use the real `cp` (GNU coreutils) by absolute path — the exact command + // Use the real `cp` (GNU/BSD coreutils) by absolute path — the exact command // installCertLinux runs — so the umask actually applies. Node's // fs.copyFileSync preserves the source mode, which would mask the bug, and // the bare `cp` on PATH below is a logging stub from the install tests. - execFileSync("/usr/bin/cp", [src, dst]); + // macOS keeps coreutils at /bin/cp; Linux (GNU coreutils) at /usr/bin/cp. + const realCp = ["/usr/bin/cp", "/bin/cp"].find((p) => fs.existsSync(p)); + assert.ok(realCp, "a real cp binary must exist for this filesystem proof"); + execFileSync(realCp, [src, dst]); const mode = fs.statSync(dst).mode & 0o777; assert.equal(mode, 0o600, "cp under umask 0077 must produce 0600 — the bug this fix repairs"); } finally { diff --git a/tests/unit/opencode-empty-rejection-rotation.test.ts b/tests/unit/opencode-empty-rejection-rotation.test.ts new file mode 100644 index 00000000000..784a4b83a74 --- /dev/null +++ b/tests/unit/opencode-empty-rejection-rotation.test.ts @@ -0,0 +1,416 @@ +import { describe, it, beforeEach, afterEach, before, after } from "node:test"; +import assert from "node:assert"; +import net from "node:net"; +import { OpencodeExecutor } from "../../open-sse/executors/opencode.ts"; +import type { ExecutorLog, ProviderCredentials } from "../../open-sse/executors/base.ts"; +import { resolveProxyForRequest } from "../../open-sse/utils/proxyFetch.ts"; +import { + isEmptyUpstreamRejection, + extractChatcmplId, +} from "../../open-sse/executors/accountRotation.ts"; + +/** + * Empty-upstream-rejection rotation (#design opencode-empty-rejection-rotation). + * + * An upstream 400 whose body carries no usable completion (the observed malformed + * envelope: `choices[0].message` with no error field, no real content, + * `finish_reason: null`) must be rotated/retried instead of propagated as a fatal + * success — that was killing subagent sessions. These tests pin the wiring: + * + * 1. A 400 empty rejection rotates to the next account (and its proxy). + * 2. The retry budget is bounded: +1 attempt for a single account, exactly N + * for an N-account all-empty run (propagate the last 400, never loop forever). + * 3. A 400 carrying a real error field (or non-empty content) still propagates + * immediately — no cooldown, no success, no rotation. + * 4. The 200/success path is never cloned or read (anti-bufferisation). + * + * The dispatch layer is mocked by stubbing globalThis.fetch (exactly what the + * #4954 proxy integration test does). Three throwaway TCP listeners stand in for + * the per-account proxies so runWithProxyContext's reachability probe passes. + */ + +const log: ExecutorLog = { debug() {}, info() {}, warn() {}, error() {} }; + +const ACCOUNT_A = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; +const ACCOUNT_B = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; +const ACCOUNT_C = "cccccccccccccccccccccccccccccccc"; + +const EMPTY_BODY = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; +const ERROR_BODY = JSON.stringify({ + error: { message: "bad request", type: "invalid_request_error" }, +}); + +let serverA: net.Server; +let serverB: net.Server; +let serverC: net.Server; +let portA = 0; +let portB = 0; +let portC = 0; + +function listen(server: net.Server): Promise { + return new Promise((resolve) => { + server.listen(0, "127.0.0.1", () => { + resolve((server.address() as net.AddressInfo).port); + }); + }); +} + +before(async () => { + serverA = net.createServer((s) => s.destroy()); + serverB = net.createServer((s) => s.destroy()); + serverC = net.createServer((s) => s.destroy()); + portA = await listen(serverA); + portB = await listen(serverB); + portC = await listen(serverC); +}); + +after(() => { + serverA?.close(); + serverB?.close(); + serverC?.close(); +}); + +function portFor(fp: string): number { + if (fp === ACCOUNT_A) return portA; + if (fp === ACCOUNT_B) return portB; + return portC; +} + +/** `fingerprints` accounts; `proxied` is the subset that get a dedicated proxy + * (defaults to all). A proxy-less account shares the default egress. */ +function credentialsFor( + fingerprints: string[], + proxied: string[] = [...fingerprints] +): ProviderCredentials { + return { + apiKey: null, + accessToken: null, + connectionId: "noauth", + providerSpecificData: { + fingerprints, + ...(proxied.length > 0 && { + accountProxies: proxied.map((fp) => ({ + fingerprint: fp, + proxy: { type: "http", host: "127.0.0.1", port: portFor(fp) }, + })), + }), + }, + }; +} + +/** A Response subclass that counts clone() so we can assert the executor never + * buffers a 200/streaming response. Note: `clone()` returns a plain Response, so + * only `clone()` is reliably counted (a read on the clone hits the native + * method, not this override) — counting clones is the meaningful invariant. */ +class SpyResponse extends Response { + static clones = 0; + clone(): Response { + SpyResponse.clones++; + return super.clone(); + } +} + +interface PlanStep { + status: number; + body?: string; + throw?: Error; +} + +describe("OpencodeExecutor empty-rejection rotation", () => { + let originalFetch: typeof globalThis.fetch; + let observed: Array<{ source: string; host: string | null; port: string | null }>; + const GUARD_FLAG = "NETWORK_ROTATION_SHARED_EGRESS_GUARD"; + let savedGuardFlag: string | undefined; + + beforeEach(() => { + originalFetch = globalThis.fetch; + observed = []; + SpyResponse.clones = 0; + savedGuardFlag = process.env[GUARD_FLAG]; + delete process.env[GUARD_FLAG]; + }); + + afterEach(() => { + globalThis.fetch = originalFetch; + if (savedGuardFlag === undefined) delete process.env[GUARD_FLAG]; + else process.env[GUARD_FLAG] = savedGuardFlag; + }); + + function installFetch(plan: PlanStep[]) { + let call = 0; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = + typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + const resolved = resolveProxyForRequest(url); + observed.push({ + source: resolved.source, + host: resolved.proxyUrl ? new URL(resolved.proxyUrl).hostname : null, + port: resolved.proxyUrl ? new URL(resolved.proxyUrl).port : null, + }); + const step = plan[Math.min(call, plan.length - 1)]; + call++; + if (step.throw) throw step.throw; + return new SpyResponse(step.body ?? JSON.stringify({ ok: step.status === 200 }), { + status: step.status, + headers: { "Content-Type": "application/json" }, + }); + }) as typeof globalThis.fetch; + } + + /** + * Launches the executor. Asserts the predicate itself behaves (regression guard + * for the design's signature — the wiring tests below depend on it). + */ + it("predicate matches the observed envelope and rejects real errors", () => { + assert.strictEqual(isEmptyUpstreamRejection(400, EMPTY_BODY), true); + assert.strictEqual(isEmptyUpstreamRejection(200, EMPTY_BODY), false); + assert.strictEqual(isEmptyUpstreamRejection(400, ERROR_BODY), false); + assert.strictEqual(extractChatcmplId(EMPTY_BODY), "chatcmpl_44fn2g6e7kk"); + }); + + it("rotates to the next account on an empty 400 rejection (loop)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 200, + "must rotate past the empty 400" + ); + assert.ok(observed.length >= 2, "should have dispatched on a second account"); + assert.ok( + observed.some((o) => o.port === String(portA)), + "first attempt on account A" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "rotated attempt on account B" + ); + }); + + it("caps an all-empty N-account run at N attempts and propagates the last 400 intact", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + { status: 200 }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 400, + "must propagate the last empty 400" + ); + assert.strictEqual(observed.length, 3, "must NOT exceed N attempts (no infinite loop)"); + assert.ok(SpyResponse.clones >= 1, "the empty 400 path must read the body to classify it"); + const propagated = await (result as { response: Response }).response.clone().text(); + assert.strictEqual(propagated, EMPTY_BODY, "propagated 400 body must stay intact"); + for (const p of observed) { + assert.strictEqual(p.source, "context", "every dispatch must egress through a proxy context"); + } + }); + + it("retries the same proxied account once when it is the only account", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A]), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + assert.strictEqual(observed.length, 2, "exactly one bounded retry on the sole account"); + assert.ok( + observed.every((o) => o.port === String(portA)), + "both attempts egress through the single account's proxy" + ); + }); + + it("coexists with 429 rotation and 200 success in the same request", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 429 }, { status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 200, + "final response should succeed" + ); + assert.strictEqual(observed.length, 3, "429 + empty-400 + success across three accounts"); + assert.ok( + observed.some((o) => o.port === String(portA)), + "account A (429)" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "account B (empty 400)" + ); + assert.ok( + observed.some((o) => o.port === String(portC)), + "account C (200)" + ); + }); + + it("propagates a 400 carrying an error field immediately (no rotation)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: ERROR_BODY }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 400, + "real error 400 must propagate" + ); + assert.strictEqual(observed.length, 1, "must NOT rotate on a genuine error 400"); + }); + + it("never clones or reads the body of a 200 via the loop", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 200 }, { status: 200 }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 200); + assert.strictEqual(SpyResponse.clones, 0, "loop 200 must never be cloned"); + }); + + it("retries once via the fast path when a direct account answers an empty 400", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + assert.strictEqual(observed.length, 2, "fast path must retry the direct account exactly once"); + }); + + it("propagates the second 400 intact when the fast path retries and empty-rejects again", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + const propagated = await (result as { response: Response }).response.clone().text(); + assert.strictEqual(propagated, EMPTY_BODY, "second rejection propagates with intact body"); + assert.strictEqual(observed.length, 2, "exactly one retry, no loop"); + }); + + it("never clones or reads the body of a 200 via the fast path", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 200); + assert.strictEqual(SpyResponse.clones, 0, "fast path 200 must never be cloned"); + }); + + it("rotates to a proxied account after a proxy-less account empty-rejects (shared-egress guard on by default)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + // A proxy-less, B proxied: B must still be tried and succeed. + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B], [ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: { status: number } }).response.status, + 200, + "the proxied account (B) must still be tried and must succeed" + ); + assert.strictEqual(observed.length, 2, "exactly one empty rejection (A) then one success (B)"); + assert.ok( + observed.some((o) => o.source === "direct"), + "first dispatch on the proxy-less account" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "rotated dispatch on the proxied account" + ); + }); +}); diff --git a/tests/unit/openrouter-free-model-credits-exhausted.test.ts b/tests/unit/openrouter-free-model-credits-exhausted.test.ts index 601553b658b..129d2fac6d9 100644 --- a/tests/unit/openrouter-free-model-credits-exhausted.test.ts +++ b/tests/unit/openrouter-free-model-credits-exhausted.test.ts @@ -78,7 +78,11 @@ test("getProviderCredentials still refuses a PAID OpenRouter model on a credits_ "anthropic/claude-opus-4.5" ); - assert.equal(selected, null, "paid-model requests must still be blocked on the exhausted connection"); + assert.deepEqual( + selected, + { allExpired: true, expiredCount: 1, expiredStatus: "credits_exhausted" }, + "paid-model requests must still be blocked on the exhausted connection" + ); }); test("getProviderCredentials still refuses a :free OpenRouter model on a banned connection", async () => { @@ -99,9 +103,9 @@ test("getProviderCredentials still refuses a :free OpenRouter model on a banned "meta-llama/llama-3.1-8b-instruct:free" ); - assert.equal( + assert.deepEqual( selected, - null, + { allExpired: true, expiredCount: 1, expiredStatus: "banned" }, "the free-model exemption only applies to credits_exhausted, not other terminal statuses" ); }); @@ -119,9 +123,9 @@ test("getProviderCredentials still refuses a :free model on a credits_exhausted const selected = await auth.getProviderCredentials("openai", null, null, "some-model:free"); - assert.equal( + assert.deepEqual( selected, - null, + { allExpired: true, expiredCount: 1, expiredStatus: "credits_exhausted" }, "the exemption is OpenRouter-specific, since only OpenRouter uses the :free naming convention with a shared balance" ); }); diff --git a/tests/unit/perf-a-b-c-d.test.ts b/tests/unit/perf-a-b-c-d.test.ts new file mode 100644 index 00000000000..0a131e0a34d --- /dev/null +++ b/tests/unit/perf-a-b-c-d.test.ts @@ -0,0 +1,26 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { logProxyEvent, flushProxyLogsSync } from "../../src/lib/proxyLogger.ts"; +import { shouldStripCloudCodeThinking } from "../../open-sse/services/cloudCodeThinking.ts"; + +test("Part A: async proxy log batching queues entries without synchronous failure", () => { + const sampleLog = { + status: "success", + provider: "test-provider", + latencyMs: 15, + }; + + const logged = logProxyEvent(sampleLog); + assert.equal(logged.provider, "test-provider"); + assert.equal(typeof logged.id, "string"); + + // Ensure flush completes without throwing + assert.doesNotThrow(() => { + flushProxyLogsSync(); + }); +}); + +test("Part D: pre-compiled regex in cloudCodeThinking model normalization", () => { + assert.equal(shouldStripCloudCodeThinking("antigravity", "antigravity/claude-3-7-sonnet"), true); + assert.equal(shouldStripCloudCodeThinking("antigravity", "models/gemini-2.5-pro"), false); +}); diff --git a/tests/unit/provider-sweep-live-discovery.test.ts b/tests/unit/provider-sweep-live-discovery.test.ts index 323a1599f3e..2431179fa7a 100644 --- a/tests/unit/provider-sweep-live-discovery.test.ts +++ b/tests/unit/provider-sweep-live-discovery.test.ts @@ -48,7 +48,7 @@ interface ModelsBody { } // provider → the upstream /models URL the route resolves from its registry baseUrl. -const LIVE_CASES: Array<{ provider: string; liveUrl: string }> = [ +const LIVE_CASES: Array<{ provider: string; liveUrl: string; source?: string }> = [ { provider: "venice", liveUrl: "https://api.venice.ai/api/v1/models" }, { provider: "deepinfra", liveUrl: "https://api.deepinfra.com/v1/openai/models" }, { provider: "wandb", liveUrl: "https://api.inference.wandb.ai/v1/models" }, @@ -62,7 +62,11 @@ const LIVE_CASES: Array<{ provider: string; liveUrl: string }> = [ { provider: "ovhcloud", liveUrl: "https://oai.endpoints.kepler.ai.cloud.ovh.net/v1/models" }, { provider: "sambanova", liveUrl: "https://api.sambanova.ai/v1/models" }, { provider: "orcarouter", liveUrl: "https://api.orcarouter.ai/v1/models" }, - { provider: "uncloseai", liveUrl: "https://hermes.ai.unturf.com/v1/models" }, + { + provider: "uncloseai", + liveUrl: "https://hermes.ai.unturf.com/v1/models", + source: "upstream", + }, { provider: "opencode-go", liveUrl: "https://opencode.ai/zen/go/v1/models" }, { provider: "baseten", liveUrl: "https://inference.baseten.co/v1/models" }, { provider: "hyperbolic", liveUrl: "https://api.hyperbolic.xyz/v1/models" }, @@ -76,7 +80,7 @@ const LIVE_CASES: Array<{ provider: string; liveUrl: string }> = [ { provider: "api-airforce", liveUrl: "https://api.airforce/v1/models" }, ]; -for (const { provider, liveUrl } of LIVE_CASES) { +for (const { provider, liveUrl, source = "api" } of LIVE_CASES) { test(`sweep: ${provider} import fetches the live /models catalog`, async () => { await resetStorage(); const connection = await providersDb.createProviderConnection({ @@ -109,7 +113,7 @@ for (const { provider, liveUrl } of LIVE_CASES) { const body = (await response.json()) as ModelsBody; assert.equal(body.provider, provider); assert.ok(fetched, `should have probed ${liveUrl}`); - assert.equal(body.source, "api", "should serve the live upstream catalog, not local_catalog"); + assert.equal(body.source, source, "should serve the live upstream catalog, not local_catalog"); const ids = body.models.map((m) => m.id); assert.ok( ids.includes(`${provider}-live-a`) && ids.includes(`${provider}-live-b`), diff --git a/tests/unit/providers-constants-split.test.ts b/tests/unit/providers-constants-split.test.ts index 3eabac24bb9..e22b3fd4e83 100644 --- a/tests/unit/providers-constants-split.test.ts +++ b/tests/unit/providers-constants-split.test.ts @@ -25,7 +25,7 @@ // #10729) brings it to 229; Token Kiosk (gateways, #10722) — merged in the same // merge-train batch — independently bumped the gateways family too, landing at 231; Freebuff // (gateways, #10531) brings it to 232. #8864 moves uncloseai (gateways family) into -// NOAUTH_PROVIDERS, dropping the APIKEY_PROVIDERS count to 231. +// NOAUTH_PROVIDERS, dropping the APIKEY_PROVIDERS count to 231. Logfare (gateways, #10987) brings it back to 232. import { test } from "node:test"; import assert from "node:assert/strict"; @@ -54,12 +54,12 @@ test("barrel still exports every catalog + key helpers", () => { } }); -test("APIKEY_PROVIDERS merges the 6 family files into 231 entries (no loss / no dup)", async () => { +test("APIKEY_PROVIDERS merges the 6 family files into 232 entries (no loss / no dup)", async () => { const keys = Object.keys((P as Record).APIKEY_PROVIDERS); - assert.equal(keys.length, 231); - assert.equal(new Set(keys).size, 231, "duplicate keys after spread-merge"); + assert.equal(keys.length, 232); + assert.equal(new Set(keys).size, 232, "duplicate keys after spread-merge"); // the merged object's entry-count equals the sum of the 6 semantic family files; families are a - // strict partition (every provider in exactly one), so the sum must be exactly 231. + // strict partition (every provider in exactly one), so the sum must be exactly 232. const families: [string, string][] = [ ["gateways", "APIKEY_PROVIDERS_GATEWAYS"], ["frontier-labs", "APIKEY_PROVIDERS_FRONTIER"], @@ -79,7 +79,7 @@ test("APIKEY_PROVIDERS merges the 6 family files into 231 entries (no loss / no seen.add(k); } } - assert.equal(famTotal, 231, "families must partition all 231 providers"); + assert.equal(famTotal, 232, "families must partition all 232 providers"); }); test("AI_PROVIDERS Proxy aggregates all sections; lookups resolve", () => { diff --git a/tests/unit/proxy-fetch-dns-retry-10443.test.ts b/tests/unit/proxy-fetch-dns-retry-10443.test.ts new file mode 100644 index 00000000000..2518f58141a --- /dev/null +++ b/tests/unit/proxy-fetch-dns-retry-10443.test.ts @@ -0,0 +1,27 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; + +test("proxyFetch identifies transient DNS and network errors (EAI_AGAIN, ENOTFOUND, ECONNREFUSED) as retryable dispatcher errors", () => { + const isRetryableError = (err: unknown): boolean => { + const msg = err instanceof Error ? err.message : String(err); + const errCode = (err as { code?: unknown })?.code; + return Boolean( + msg.includes("fetch failed") || + errCode === "ECONNREFUSED" || + msg.includes("ECONNREFUSED") || + errCode === "EAI_AGAIN" || + msg.includes("EAI_AGAIN") || + errCode === "ENOTFOUND" || + msg.includes("ENOTFOUND") || + errCode === "ETIMEDOUT" || + msg.includes("ETIMEDOUT") || + (typeof errCode === "string" && errCode.startsWith("UND_ERR")) || + msg.includes("UND_ERR") + ); + }; + + assert.equal(isRetryableError({ code: "EAI_AGAIN", message: "getaddrinfo EAI_AGAIN www.googleapis.com" }), true); + assert.equal(isRetryableError({ code: "ENOTFOUND", message: "getaddrinfo ENOTFOUND api.example.com" }), true); + assert.equal(isRetryableError({ code: "ECONNREFUSED", message: "connect ECONNREFUSED 127.0.0.1:20128" }), true); + assert.equal(isRetryableError(new Error("HTTP 404 Not Found")), false); +}); diff --git a/tests/unit/rateLimitManager-mintime-floor-9763.test.ts b/tests/unit/rateLimitManager-mintime-floor-9763.test.ts new file mode 100644 index 00000000000..2c4ee557552 --- /dev/null +++ b/tests/unit/rateLimitManager-mintime-floor-9763.test.ts @@ -0,0 +1,88 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const rlm = await import("../../open-sse/services/rateLimitManager.ts"); +const { + enableRateLimitProtection, + withRateLimit, + updateFromHeaders, + applyRequestQueueSettings, + __setLimiterFactoryForTests, + __resetRateLimitManagerForTests, +} = rlm; + +test.beforeEach(async () => { + await __resetRateLimitManagerForTests(); +}); + +test("headroom relaxation respects operator minTimeBetweenRequestsMs floor (#9763)", async () => { + // Apply an operator-configured minTime floor of 200ms + await applyRequestQueueSettings({ + minTimeBetweenRequestsMs: 200, + concurrentRequests: 0, + requestsPerMinute: 0, + maxWaitMs: 30000, + autoEnableApiKeyProvider: false, + }); + + let capturedMinTime: number | undefined; + + // Inject a fake limiter whose updateSettings captures the minTime. + // eslint-disable-next-line @typescript-eslint/no-explicit-any + const noop = (): any => undefined; + + __setLimiterFactoryForTests(() => { + const listeners: Record void>> = {}; + const fake = { + // eslint-disable-next-line @typescript-eslint/no-explicit-any + updateSettings(updates: Record) { + capturedMinTime = typeof updates.minTime === "number" ? updates.minTime : undefined; + return fake; + }, + on(event: string, fn: (...args: unknown[]) => void) { + (listeners[event] ??= []).push(fn); + return fake; + }, + schedule(arg0: unknown, arg1?: unknown) { + const fn = typeof arg1 === "function" ? arg1 : typeof arg0 === "function" ? arg0 : noop; + return fn(); + }, + disconnect() { + return Promise.resolve(); + }, + chain() { + return fake; + }, + counts() { + return { RECEIVED: 0, QUEUED: 0, RUNNING: 0, EXECUTING: 0 }; + }, + currentReservoir() { + return Promise.resolve(null); + }, + stop() { + return Promise.resolve(); + }, + }; + return fake; + }); + + enableRateLimitProtection("test-mintime-floor"); + + // Materialize the limiter with a dummy request + await withRateLimit("openai", "test-mintime-floor", "gpt-4", async () => "ok"); + + // Simulate a response with plenty of headroom: remaining=80 > limit*0.5=50 + const headers = new Headers({ + "x-ratelimit-limit-requests": "100", + "x-ratelimit-remaining-requests": "80", + }); + updateFromHeaders("openai", "test-mintime-floor", headers, 200, "gpt-4"); + + // The operator configured minTime=200, so headroom relaxation MUST NOT + // override it to 0. Before the fix, capturedMinTime === 0 (RED). + assert.strictEqual( + capturedMinTime, + 200, + `Expected minTime=200 (operator floor), got ${capturedMinTime}` + ); +}); diff --git a/tests/unit/readyz-route.test.ts b/tests/unit/readyz-route.test.ts index ec9ed1cb12b..ef5c0f67ee5 100644 --- a/tests/unit/readyz-route.test.ts +++ b/tests/unit/readyz-route.test.ts @@ -49,9 +49,11 @@ test("/readyz is omitted from the centralized auth proxy matcher", () => { assert.equal(/["']\/healthz/.test(matcherBlock), false); }); -test("/readyz re-exports the /healthz handlers (no second lifecycle)", () => { +test("/readyz re-exports the /healthz handlers and declares its route config locally", () => { const source = fs.readFileSync("src/app/readyz/route.ts", "utf8"); assert.match(source, /from ["']\.\.\/healthz\/route["']/); + assert.match(source, /export const dynamic = ["']force-dynamic["']/); + assert.doesNotMatch(source, /export\s*\{[^}]*\bdynamic\b[^}]*\}\s*from/); assert.equal(/monitoring/i.test(source), false); assert.equal(/sqlite/i.test(source), false); }); diff --git a/tests/unit/reasoning-cache.test.ts b/tests/unit/reasoning-cache.test.ts index aded469ca85..db538a80314 100644 --- a/tests/unit/reasoning-cache.test.ts +++ b/tests/unit/reasoning-cache.test.ts @@ -752,47 +752,70 @@ describe("Reasoning Replay Cache — Translator Replay", () => { assert.equal(lookupReasoning(callId), "Authentic provider reasoning"); }); - it("should never cache Responses summaries or opaque plaintext companions", () => { - for (const [suffix, reasoningItem] of [ - [ - "summary", - { - type: "reasoning", - summary: [{ type: "summary_text", text: "Display-only summary" }], - }, - ], - [ - "mixed", - { - type: "reasoning", - encrypted_content: "opaque-provider-state", - content: [{ type: "reasoning_text", text: "Unsafe plaintext companion" }], - summary: [{ type: "summary_text", text: "Display-only mixed summary" }], - }, - ], - ] as const) { - clearReasoningCacheAll(); - const callId = `call_nonstream_${suffix}_reasoning`; - const translated = translateNonStreamingResponse( - { - object: "response", - model: "deepseek-v4-flash", - output: [ - reasoningItem, - { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, - ], - }, - FORMATS.OPENAI_RESPONSES, - FORMATS.OPENAI - ) as { choices?: Array<{ message?: Record }> }; - const message = translated.choices?.[0]?.message; - - assert.ok(message); - assert.equal(message.reasoning_content, undefined); - assert.ok(Array.isArray(message.reasoning_summary)); - assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 0); - assert.equal(lookupReasoning(callId), null); - } + it("preserves plaintext reasoning from a mixed plaintext + encrypted_content item (#10949)", () => { + clearReasoningCacheAll(); + const callId = "call_nonstream_mixed_reasoning"; + const translated = translateNonStreamingResponse( + { + object: "response", + model: "deepseek-v4-flash", + output: [ + { + type: "reasoning", + content: [ + { + type: "reasoning_text", + text: "Let me start by reading the directory to understand the structure of the corpus.", + }, + ], + encrypted_content: "", + summary: [], + }, + { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, + ], + }, + FORMATS.OPENAI_RESPONSES, + FORMATS.OPENAI + ) as { choices?: Array<{ message?: Record }> }; + const message = translated.choices?.[0]?.message; + + assert.ok(message); + assert.equal( + message.reasoning_content, + "Let me start by reading the directory to understand the structure of the corpus." + ); + assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 1); + assert.equal( + lookupReasoning(callId), + "Let me start by reading the directory to understand the structure of the corpus." + ); + }); + + it("should never cache summary-only Responses reasoning", () => { + clearReasoningCacheAll(); + const callId = "call_nonstream_summary_reasoning"; + const translated = translateNonStreamingResponse( + { + object: "response", + model: "deepseek-v4-flash", + output: [ + { + type: "reasoning", + summary: [{ type: "summary_text", text: "Display-only summary" }], + }, + { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, + ], + }, + FORMATS.OPENAI_RESPONSES, + FORMATS.OPENAI + ) as { choices?: Array<{ message?: Record }> }; + const message = translated.choices?.[0]?.message; + + assert.ok(message); + assert.equal(message.reasoning_content, undefined); + assert.ok(Array.isArray(message.reasoning_summary)); + assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 0); + assert.equal(lookupReasoning(callId), null); }); it("should preserve client-provided reasoning content", () => { diff --git a/tests/unit/rejected-request-usage.test.ts b/tests/unit/rejected-request-usage.test.ts index a25ada202b2..300e77e013c 100644 --- a/tests/unit/rejected-request-usage.test.ts +++ b/tests/unit/rejected-request-usage.test.ts @@ -60,12 +60,17 @@ test("gate-rejected request is attributed to the api key in usage_history", asyn assert.equal(keyRows.length, 1, "expected one usage_history row for the rejected request"); assert.equal(keyRows[0].success, false, "rejected request must be recorded as success:false"); - // call_logs visibility is preserved (dashboard/logs). - const logs = await callLogs.getCallLogs({}); - const rejected = (logs.logs ?? logs).filter?.( - (l: { apiKeyName?: string | null }) => l.apiKeyName === "opencode-mac" - ); - assert.ok(rejected && rejected.length >= 1, "expected a call_logs row for the rejected request"); + // call_logs visibility is preserved (dashboard/logs). saveCallLog is + // fire-and-forget inside recordRejectedRequestUsage, so poll briefly for the + // row instead of asserting synchronously after the await. + let rejected: Array<{ apiKeyName?: string | null }> = []; + for (let i = 0; i < 50 && rejected.length === 0; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + rejected = (list ?? []).filter((l) => l.apiKeyName === "opencode-mac"); + if (rejected.length === 0) await new Promise((r) => setTimeout(r, 10)); + } + assert.ok(rejected.length >= 1, "expected a call_logs row for the rejected request"); }); test("combo-exhausted rejection is also counted per api key", async () => { @@ -111,10 +116,15 @@ test("combo-exhausted rejection persists the client request body for dashboard i requestBody: { model: "default", messages: [{ role: "user", content: "hello" }] }, }); - const logs = await callLogs.getCallLogs({}); - const rejected = (logs.logs ?? logs).find?.( - (l: { apiKeyName?: string | null }) => l.apiKeyName === "request-body-test" - ); + // saveCallLog is fire-and-forget — poll briefly for the row. + let rejected: { id: string; hasRequestBody: boolean } | undefined; + for (let i = 0; i < 50 && !rejected; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + const found = (list ?? []).find((l) => l.apiKeyName === "request-body-test"); + if (found) rejected = found as unknown as { id: string; hasRequestBody: boolean }; + else await new Promise((r) => setTimeout(r, 10)); + } assert.ok(rejected, "expected a call_logs row for the rejected request"); assert.equal(rejected.hasRequestBody, true, "expected hasRequestBody to be true"); @@ -140,10 +150,15 @@ test("combo-exhausted rejection without a request body still logs cleanly (no re startTime: Date.now() - 100, }); - const logs = await callLogs.getCallLogs({}); - const rejected = (logs.logs ?? logs).find?.( - (l: { apiKeyName?: string | null }) => l.apiKeyName === "no-body-test" - ); + // saveCallLog is fire-and-forget — poll briefly for the row. + let rejected: { id: string } | undefined; + for (let i = 0; i < 50 && !rejected; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + const found = (list ?? []).find((l) => l.apiKeyName === "no-body-test"); + if (found) rejected = found as unknown as { id: string }; + else await new Promise((r) => setTimeout(r, 10)); + } assert.ok(rejected, "expected a call_logs row even without a request body"); assert.equal(rejected.hasRequestBody, false); }); diff --git a/tests/unit/security-route-guard-tiers.test.ts b/tests/unit/security-route-guard-tiers.test.ts new file mode 100644 index 00000000000..dfc1f3c026f --- /dev/null +++ b/tests/unit/security-route-guard-tiers.test.ts @@ -0,0 +1,10 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { isLocalOnlyPath } from "../../src/server/authz/routeGuard.ts"; + +test("isLocalOnlyPath correctly classifies process-spawning endpoints under Tier 1 LOCAL_ONLY", () => { + assert.equal(isLocalOnlyPath("/api/services/dario/start"), true); + assert.equal(isLocalOnlyPath("/api/mcp/stream"), true); + assert.equal(isLocalOnlyPath("/api/cli-tools/runtime/status"), true); + assert.equal(isLocalOnlyPath("/api/v1/chat/completions"), false); +}); diff --git a/tests/unit/services/ServiceSupervisor.test.ts b/tests/unit/services/ServiceSupervisor.test.ts index 011395733ad..6e04455e614 100644 --- a/tests/unit/services/ServiceSupervisor.test.ts +++ b/tests/unit/services/ServiceSupervisor.test.ts @@ -18,6 +18,9 @@ const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-superviso process.env.DATA_DIR = TEST_DATA_DIR; process.env.NODE_ENV = "test"; process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; +// Adoption is intentionally opt-in after GHSA-wg9p-6m2g-4v27. These tests +// exercise the explicit adoption path, so enable it for this isolated process. +process.env.OMNIROUTE_ADOPT_EXISTING_SERVICE = "1"; // Import DB core first to trigger migration (creates version_manager with new columns) const core = await import("../../../src/lib/db/core.ts"); diff --git a/tests/unit/services/portProbePid.test.ts b/tests/unit/services/portProbePid.test.ts index b88383fd801..c58bf86045c 100644 --- a/tests/unit/services/portProbePid.test.ts +++ b/tests/unit/services/portProbePid.test.ts @@ -64,6 +64,12 @@ test("parseNetstatPid matches on the local address, not the foreign one", () => assert.equal(parseNetstatPid(stdout, 20128), 596922); }); +test("parseNetstatPid reads macOS process:pid output", () => { + const stdout = + "tcp4 0 0 127.0.0.1.20128 *.* LISTEN 0 0 131072 131072 node:596922 00100\n"; + assert.equal(parseNetstatPid(stdout, 20128), 596922); +}); + test("parseNetstatPid ignores non-listening rows and unknown ports", () => { const stdout = "tcp 0 0 127.0.0.1:20128 1.2.3.4:5555 ESTABLISHED 596922/node\n"; diff --git a/tests/unit/sse-auth-codex-account-pool.test.ts b/tests/unit/sse-auth-codex-account-pool.test.ts index 581efd1f339..927775d15f7 100644 --- a/tests/unit/sse-auth-codex-account-pool.test.ts +++ b/tests/unit/sse-auth-codex-account-pool.test.ts @@ -242,8 +242,8 @@ test("Codex parent authentication failures block both virtual children without c const inventory = await providersDb.getProviderConnections({ provider: "codex" }); assert.equal(unavailable.shouldFallback, true); - assert.equal(spark, null); - assert.equal(normal, null); + assert.deepEqual(spark, { allExpired: true, expiredCount: 1, expiredStatus: "expired" }); + assert.deepEqual(normal, { allExpired: true, expiredCount: 1, expiredStatus: "expired" }); assert.deepEqual( inventory.map((item) => item.id), [connection.id] diff --git a/tests/unit/stream-continuation-wiring.test.ts b/tests/unit/stream-continuation-wiring.test.ts index 2f246e325d4..36a6f445d1f 100644 --- a/tests/unit/stream-continuation-wiring.test.ts +++ b/tests/unit/stream-continuation-wiring.test.ts @@ -47,9 +47,15 @@ async function collectText(stream: ReadableStream): Promise const ROLE = 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n'; const content = (s: string) => `data: {"choices":[{"delta":{"content":${JSON.stringify(s)}}}]}\n\n`; +const reasoning = (s: string) => + `data: {"choices":[{"delta":{"reasoning_content":${JSON.stringify(s)}}}]}\n\n`; +const finishStopNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'; +const finishLengthNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"length"}]}\n\n'; + test("mid-stream continuation: stitches the suffix after a silent post-commit truncation", async () => { - // Commits on chunk 1, emits "Hello wor", then ends WITHOUT a terminal marker (silent cut). - const initial = streamFrom([ROLE, content("Hello wor")]); + // Commits on chunk 1, emits "Hello there world", then ends WITHOUT a terminal marker + // (silent cut). + const initial = streamFrom([ROLE, content("Hello there world")]); let finalizeCount = 0; let continueArg = ""; @@ -60,15 +66,22 @@ test("mid-stream continuation: stitches the suffix after a silent post-commit tr now: steppingClock(), continueStream: async (soFar: string) => { continueArg = soFar; - // The model re-emits a small overlap ("wor") which must be trimmed away. - return streamFrom([ROLE, content("world!"), "data: [DONE]\n\n"]); + // The model re-emits only a partial tail of what was already sent ("there world", + // 11 chars — above the 8-char threshold, but NOT the full emitted text, unlike a + // full-string overlap this stays a discriminating test of trimContinuationOverlap's + // partial-tail trim, not just its "accept everything" path) before continuing. + return streamFrom([ROLE, content("there world, nice to meet you!"), "data: [DONE]\n\n"]); }, }); const out = await collectText(stream); const scan = scanOpenAiSseText(out); - assert.equal(continueArg, "Hello wor", "continuation is prefilled with the text already sent"); - assert.equal(scan.text, "Hello world!", "client sees the full answer, overlap trimmed, exactly once"); + assert.equal(continueArg, "Hello there world", "continuation is prefilled with the text already sent"); + assert.equal( + scan.text, + "Hello there world, nice to meet you!", + "client sees the full answer, partial overlap trimmed, exactly once" + ); assert.equal(scan.terminal, true, "the recovered stream ends with a terminal marker"); assert.equal(finalizeCount, 1, "finalize runs exactly once"); }); @@ -80,7 +93,7 @@ test("mid-stream continuation: recovers a post-commit transport error too", asyn const stream = createRecoverableStream(initial, async () => null, { finalize: () => {}, now: steppingClock(), - continueStream: async () => streamFrom([content("answer done."), "data: [DONE]\n\n"]), + continueStream: async () => streamFrom([content("Partial answer done."), "data: [DONE]\n\n"]), }); const scan = scanOpenAiSseText(await collectText(stream)); assert.equal(scan.text, "Partial answer done."); @@ -115,3 +128,167 @@ test("tool-call in flight is never continued (would corrupt tool JSON)", async ( await collectText(stream); assert.equal(continued, false, "continuation must NOT fire once a tool call has started streaming"); }); + +test("mid-stream continuation: a zero-overlap restart is rejected, never concatenated raw", async () => { + // Truncates silently after real, non-empty text — canContinue() fires. + const initial = streamFrom([ROLE, content("Tous les faits sont reunis")]); + let continuations = 0; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + maxContinuations: 1, + continueStream: async () => { + continuations += 1; + // The model ignores the assistant prefill and restarts on an unrelated sentence — + // zero characters of overlap with what was already emitted. + return streamFrom([ + content("Je complete le design - derniere verification"), + "data: [DONE]\n\n", + ]); + }, + }); + const out = await collectText(stream); + const scan = scanOpenAiSseText(out); + assert.equal( + scan.text, + "Tous les faits sont reunis", + "the unrelated restart must never be appended to the already-emitted text" + ); + assert.equal(scan.terminal, true, "closes cleanly instead of leaving the client hanging"); + assert.equal(continuations, 1, "bounded by maxContinuations — does not loop forever"); +}); + +test("mid-stream continuation: a nonzero overlap below the threshold is rejected too", async () => { + // Genuine 4-character overlap ("pret"), well under the 8-char threshold — this is the + // false-negative case a naive `overlapChars === 0` check would miss (a restart that + // happens to share a short accidental fragment with the emitted tail): must still be + // treated as a suspected restart, not accepted as a genuine resume. + const initial = streamFrom([ROLE, content("Le design est pret")]); + let continuations = 0; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + maxContinuations: 1, + continueStream: async () => { + continuations += 1; + // Shares only "pret" (4 chars) with the emitted tail, then diverges completely. + return streamFrom([content("pret a partir de zero"), "data: [DONE]\n\n"]); + }, + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal( + scan.text, + "Le design est pret", + "a below-threshold (but nonzero) overlap must not be accepted as a real resume" + ); + assert.equal(continuations, 1); +}); + +test("mid-stream continuation: a real overlap at or above the threshold is still stitched correctly", async () => { + // Regression guard: the existing happy path (first test in this file, whose updated + // fixture re-emits the 11-char partial tail "there world") still passes below — this test + // adds an overlap AT the threshold boundary to prove Task 3's new check does not fire when + // it shouldn't. + const initial = streamFrom([ROLE, content("The answer to this question")]); + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => + // "question" (8 chars) overlaps the tail of emittedText exactly at the threshold. + streamFrom([content("question is forty-two."), "data: [DONE]\n\n"]), + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal( + scan.text, + "The answer to this question is forty-two.", + "an overlap meeting the threshold is trimmed and stitched, not rejected" + ); +}); + +test("mid-stream continuation: a clean stop with reasoning-only output (no answer) triggers a continuation", async () => { + const initial = streamFrom([ + ROLE, + reasoning("the model thinks through the problem here..."), + finishStopNoContent, + ]); + let continueArg = "__unset__"; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async (soFar: string) => { + continueArg = soFar; + return streamFrom([content("Here is the actual answer."), "data: [DONE]\n\n"]); + }, + }); + const out = await collectText(stream); + const scan = scanOpenAiSseText(out); + assert.equal(continueArg, "", "nothing usable was emitted — the re-request has an empty prefill"); + assert.equal( + scan.text, + "Here is the actual answer.", + "the client gets a real answer instead of silence" + ); + assert.equal(scan.terminal, true); +}); + +test("mid-stream continuation: a clean stop with truly empty output (no text, no reasoning) is left unchanged", async () => { + const initial = streamFrom([ROLE, finishStopNoContent]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + await collectText(stream); + assert.equal( + continued, + false, + "no reasoning trace means there is nothing to act on — do not guess" + ); +}); + +test("mid-stream continuation: finish_reason 'length' with reasoning-only output does NOT trigger a continuation", async () => { + // Regression guard for a blocker found in cross-review: widening the gate to any + // terminal marker (instead of the literal finish_reason "stop") would wrongly spend a + // continuation attempt on a token-limit cutoff, which is out of this fix's scope. + const initial = streamFrom([ + ROLE, + reasoning("the model was still thinking when it hit the token limit..."), + finishLengthNoContent, + ]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + await collectText(stream); + assert.equal(continued, false, "finish_reason 'length' is out of scope for this fix"); +}); + +test("mid-stream continuation: real content alongside reasoning at a clean stop is left unchanged (non-regression)", async () => { + const initial = streamFrom([ + ROLE, + reasoning("thinking..."), + content("The real answer."), + finishStopNoContent, + ]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal(continued, false, "real content was delivered — nothing to recover"); + assert.equal(scan.text, "The real answer."); +}); diff --git a/tests/unit/stream-continuation.test.ts b/tests/unit/stream-continuation.test.ts index caffe936df9..8a1da37ce9d 100644 --- a/tests/unit/stream-continuation.test.ts +++ b/tests/unit/stream-continuation.test.ts @@ -21,6 +21,30 @@ test("scanOpenAiSseText accumulates content deltas and flags an OpenAI-compat st assert.equal(r.terminal, false); }); +test("scanOpenAiSseText accumulates reasoning_content deltas separately from content", () => { + const sse = + 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n' + + 'data: {"choices":[{"delta":{"reasoning_content":"thinking..."}}]}\n\n' + + 'data: {"choices":[{"delta":{"reasoning_content":" more"}}]}\n\n'; + const r = scanOpenAiSseText(sse); + assert.equal(r.reasoningText, "thinking... more"); + assert.equal(r.text, "", "reasoning_content must never leak into the visible text field"); + assert.equal(r.parsedOpenAi, true); +}); + +test("scanOpenAiSseText captures the literal finish_reason value", () => { + const stop = scanOpenAiSseText('data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'); + assert.equal(stop.finishReason, "stop"); + + const length = scanOpenAiSseText( + 'data: {"choices":[{"delta":{"content":"x"},"finish_reason":"length"}]}\n\n' + ); + assert.equal(length.finishReason, "length"); + + const none = scanOpenAiSseText('data: {"choices":[{"delta":{"content":"x"}}]}\n\n'); + assert.equal(none.finishReason, null, "no finish_reason seen means null, not a guessed default"); +}); + test("scanOpenAiSseText detects the terminal [DONE] marker", () => { const r = scanOpenAiSseText('data: {"choices":[{"delta":{"content":"hi"}}]}\n\ndata: [DONE]\n\n'); assert.equal(r.text, "hi"); @@ -65,6 +89,15 @@ test("makeContinuationBody refuses bodies without a messages array or empty text assert.equal(makeContinuationBody(null as never, "t"), null); }); +test("makeContinuationBody accepts an empty prefill by re-sending the messages unchanged", () => { + const body = { model: "x", stream: true, messages: [{ role: "user", content: "hi" }] }; + const out = makeContinuationBody(body, ""); + assert.ok(out, "an empty prefill must still produce a re-request body, not null"); + assert.equal(out!.messages.length, 1, "no empty assistant turn is appended"); + assert.deepEqual(out!.messages[0], { role: "user", content: "hi" }); + assert.equal(out!.stream, true); +}); + // ── trimContinuationOverlap ─────────────────────────────────────────────────── test("trimContinuationOverlap removes a duplicated seam so the join is append-only", () => { diff --git a/tests/unit/stream-handler.test.ts b/tests/unit/stream-handler.test.ts index d19a13f351e..8c19c088021 100644 --- a/tests/unit/stream-handler.test.ts +++ b/tests/unit/stream-handler.test.ts @@ -167,6 +167,41 @@ test("createDisconnectAwareStream treats cancel after Responses completed as suc assert.equal(disconnectHandled, false); }); +test("createDisconnectAwareStream recognizes a large Responses compaction completion", async () => { + let errorHandled = false; + const completed = `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + status: "completed", + output: [{ type: "compaction", encrypted_content: "x".repeat(5000) }], + }, + })}\n\n`; + const transformStream = { + readable: new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(completed)); + controller.close(); + }, + }), + writable: createNoopAbortWritable(), + }; + + const stream = createDisconnectAwareStream( + transformStream, + createStreamController({ + clientResponseFormat: FORMATS.OPENAI_RESPONSES, + onError() { + errorHandled = true; + }, + }) + ); + const text = await readStreamText(stream); + + assert.equal(text, completed); + assert.equal(errorHandled, false); + assert.doesNotMatch(text, /response\.failed/); +}); + test("createDisconnectAwareStream: Gemini 503 high-demand error becomes SSE error chunk with message preserved", async () => { const geminiMsg = "[503]: This model is currently experiencing high demand. Spikes in demand are usually temporary. Please try again later."; diff --git a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts index 25eae7a839c..8b3059236dd 100644 --- a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts +++ b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts @@ -8,7 +8,7 @@ import { omitEncryptedReasoningForLog } from "../../src/lib/logPayloads.ts"; // Responses reasoning replay is target-scoped. Plaintext DeepSeek state and // provider-generated opaque state are never interchangeable. -test("unknown Responses targets reject opaque reasoning and ignore display summaries", () => { +test("unknown Responses targets drop opaque reasoning and preserve display summaries (#10959)", () => { const body: Record = { input: [ { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, @@ -24,11 +24,14 @@ test("unknown Responses targets reject opaque reasoning and ignore display summa ], }; - const originalInput = structuredClone(body.input); const result = applyReasoningInputPolicy(body, "responses"); - assert.equal(result.incompatibleReasoning, true); - assert.deepEqual(body.input, originalInput, "rejection must not mutate the request"); + assert.equal(result.incompatibleReasoning, false); + assert.deepEqual(body.input, [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "reasoning", summary: [{ text: "display only" }] }, + { type: "function_call", name: "search", arguments: "{}", call_id: "call_1" }, + ]); }); test("unannotated targets preserve plaintext Responses reasoning without synthetic IDs", () => { @@ -132,7 +135,7 @@ test("Chat drop removes opaque state while preserving plaintext and summary deta ]); }); -test("DeepSeek rejects plaintext reasoning carrying opaque provider state", () => { +test("DeepSeek projects plaintext reasoning carrying opaque provider state onto the plaintext transport (#10949)", () => { for (const opaqueField of ["signature", "format"] as const) { const body: Record = { input: [ @@ -148,13 +151,11 @@ test("DeepSeek rejects plaintext reasoning carrying opaque provider state", () = const result = applyReasoningInputPolicy(body, "responses", { provider: "deepseek" }); - assert.equal(result.incompatibleReasoning, true, opaqueField); + assert.equal(result.incompatibleReasoning, false, opaqueField); assert.deepEqual(body.input, [ { - id: "rs_mixed123", type: "reasoning", content: [{ type: "reasoning_text", text: "untrusted companion" }], - [opaqueField]: "provider-state", }, { type: "message", role: "user", content: [{ type: "input_text", text: "continue" }] }, ]); @@ -199,6 +200,49 @@ test("drop fallback removes only the incompatible active transport and preserves assert.equal(opaqueReasoning.encrypted_content, "provider-state"); }); +test("mixed plaintext + opaque reasoning follows the target transport instead of rejecting (#10949)", () => { + const mixedReasoning = { + id: "rs_mixed", + type: "reasoning", + content: [{ type: "reasoning_text", text: "inspect first" }], + encrypted_content: "provider-state", + summary: [{ type: "summary_text", text: "display only" }], + }; + + // Plaintext target (deepseek): keep the portable plaintext, strip opaque state. + const toPlaintext: Record = { + input: [structuredClone(mixedReasoning)], + }; + const plaintextResult = applyReasoningInputPolicy(toPlaintext, "responses", { + provider: "deepseek", + }); + assert.equal(plaintextResult.incompatibleReasoning, false); + assert.deepEqual(toPlaintext.input, [ + { + type: "reasoning", + content: [{ type: "reasoning_text", text: "inspect first" }], + summary: [{ type: "summary_text", text: "display only" }], + }, + ]); + + // Opaque target (openai): keep the provider state, strip the plaintext. + const toOpaque: Record = { + input: [structuredClone(mixedReasoning)], + }; + const opaqueResult = applyReasoningInputPolicy(toOpaque, "responses", { + provider: "openai", + }); + assert.equal(opaqueResult.incompatibleReasoning, false); + assert.deepEqual(toOpaque.input, [ + { + id: "rs_mixed", + type: "reasoning", + encrypted_content: "provider-state", + summary: [{ type: "summary_text", text: "display only" }], + }, + ]); +}); + test("drop fallback preserves reasoning when its transport is compatible", () => { const body: Record = { input: [ diff --git a/tests/unit/systemd-notify.test.mjs b/tests/unit/systemd-notify.test.mjs index cb9e441f8f7..b716d57d9d5 100644 --- a/tests/unit/systemd-notify.test.mjs +++ b/tests/unit/systemd-notify.test.mjs @@ -37,8 +37,10 @@ function python3Available() { } } -// Waits for the listener to emit `expected` lines (in order), then resolves -// with everything it saw. Fails loudly on timeout or premature exit. +// Waits for the listener to emit every `expected` line, then resolves with +// everything it saw. The notifier spawns one process per signal, so AF_UNIX +// datagram arrival order is not guaranteed across those processes. +// Fails loudly on timeout or premature exit. // BARRIER=1 datagrams (sd_notify synchronization emitted by the systemd-notify // CLI after every message) are noise for this contract and are skipped. function waitForLines(child, expected, timeoutMs) { @@ -58,14 +60,14 @@ function waitForLines(child, expected, timeoutMs) { buf = buf.slice(idx + 1); if (!line || line === "BARRIER=1") continue; seen.push(line); - if (seen.length === expected.length) { + if (expected.every((expectedLine) => seen.includes(expectedLine))) { clearTimeout(timer); resolve([...seen]); } } }); child.on("exit", () => { - if (seen.length < expected.length) { + if (expected.some((expectedLine) => !seen.includes(expectedLine))) { clearTimeout(timer); reject(new Error(`listener exited early; got: ${seen.join(", ")}`)); } @@ -248,7 +250,7 @@ test( notifier.watchdog(); notifier.stopping(); const received = await waitForLines(listener, ["READY=1", "WATCHDOG=1", "STOPPING=1"], 10000); - assert.deepEqual(received, ["READY=1", "WATCHDOG=1", "STOPPING=1"]); + assert.deepEqual(received.toSorted(), ["READY=1", "STOPPING=1", "WATCHDOG=1"]); notifier.dispose(); } finally { listener.kill(); diff --git a/tests/unit/terminal-status-origin.test.ts b/tests/unit/terminal-status-origin.test.ts index a4aacf767fc..3a9221f2cab 100644 --- a/tests/unit/terminal-status-origin.test.ts +++ b/tests/unit/terminal-status-origin.test.ts @@ -1,6 +1,8 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; const DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-t11-")); process.env.DATA_DIR = DIR; @@ -9,15 +11,42 @@ const { createProviderConnection } = await import("../../src/lib/db/providers.ts const { runAsProbe } = await import("../../src/shared/utils/probeOrigin.ts"); const { writeTerminalStatus } = await import("../../src/shared/utils/terminalStatus.ts"); -test.after(() => { core.resetDbInstance(); fs.rmSync(DIR, {recursive:true, force:true}); }); +test.after(() => { + core.resetDbInstance(); + fs.rmSync(DIR, { recursive: true, force: true }); +}); -function row(id: string){ return (core.getDbInstance() as unknown as Record).prepare("SELECT is_active, test_status FROM provider_connections WHERE id=?").get(id); } +function row(id: string): { is_active: number; test_status: string } { + const result = core + .getDbInstance() + .prepare("SELECT is_active, test_status FROM provider_connections WHERE id=?") + .get(id); + assert.ok(result && typeof result === "object"); + return result as { is_active: number; test_status: string }; +} test("probe-origin writeTerminalStatus records error but never deactivates", async () => { - const conn = await createProviderConnection({ provider:"openai", authType:"apikey", name:"t11", apiKey:"sk-t11", isActive:true, testStatus:"active" } as unknown as Record); - const id = String((conn as unknown as Record).id); + const conn = await createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "t11", + apiKey: "sk-t11", + isActive: true, + testStatus: "active", + }); + const id = String(conn.id); await runAsProbe(async () => { - await writeTerminalStatus(id, { testStatus:"banned", isActive:false, lastError:"probe 403", errorCode:"403", lastErrorType:"FORBIDDEN" }, "probe"); + await writeTerminalStatus( + id, + { + testStatus: "banned", + isActive: false, + lastError: "probe 403", + errorCode: "403", + lastErrorType: "FORBIDDEN", + }, + "probe" + ); }); const r = row(id); assert.equal(r.is_active, 1); // probe n'a jamais désactivé @@ -25,9 +54,26 @@ test("probe-origin writeTerminalStatus records error but never deactivates", asy }); test("production writeTerminalStatus deactivates on terminal", async () => { - const conn = await createProviderConnection({ provider:"openai", authType:"apikey", name:"t11b", apiKey:"sk-t11b", isActive:true, testStatus:"active" } as unknown as Record); - const id = String((conn as unknown as Record).id); - await writeTerminalStatus(id, { testStatus:"banned", isActive:false, lastError:"real 403", errorCode:"403", lastErrorType:"FORBIDDEN" }, "production"); + const conn = await createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "t11b", + apiKey: "sk-t11b", + isActive: true, + testStatus: "active", + }); + const id = String(conn.id); + await writeTerminalStatus( + id, + { + testStatus: "banned", + isActive: false, + lastError: "real 403", + errorCode: "403", + lastErrorType: "FORBIDDEN", + }, + "production" + ); const r = row(id); assert.equal(r.is_active, 0); assert.equal(r.test_status, "banned"); diff --git a/tests/unit/translator-openai-responses-req.test.ts b/tests/unit/translator-openai-responses-req.test.ts index da87b26116f..2a05f57310e 100644 --- a/tests/unit/translator-openai-responses-req.test.ts +++ b/tests/unit/translator-openai-responses-req.test.ts @@ -172,28 +172,26 @@ test("Responses -> Chat keeps summary-only reasoning out of continuation state", assert.equal(result.messages[0].reasoning_content, undefined); }); -test("Responses -> Chat rejects opaque reasoning instead of replaying its plaintext companion", () => { - assert.throws( - () => - openaiResponsesToOpenAIRequest( - "deepseek-v4-pro", +test("Responses -> Chat replays the plaintext companion of an opaque reasoning item (#10949)", () => { + const result = openaiResponsesToOpenAIRequest( + "deepseek-v4-pro", + { + input: [ { - input: [ - { - id: "rs_opaque", - type: "reasoning", - encrypted_content: "opaque-provider-state", - content: [{ type: "reasoning_text", text: "Untrusted plaintext companion" }], - summary: [{ type: "summary_text", text: "Display summary" }], - }, - { type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, - ], + id: "rs_opaque", + type: "reasoning", + encrypted_content: "opaque-provider-state", + content: [{ type: "reasoning_text", text: "Untrusted plaintext companion" }], + summary: [{ type: "summary_text", text: "Display summary" }], }, - false, - { _preserveReasoningContent: true } - ), - /Reasoning continuation is not compatible/ - ); + { type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, + ], + }, + false, + { _preserveReasoningContent: true } + ) as { messages: Array> }; + + assert.equal(result.messages[0].reasoning_content, "Untrusted plaintext companion"); }); test("Responses -> Chat merges assistant text that follows a function call", () => { diff --git a/tests/unit/translator-resp-openai-responses.test.ts b/tests/unit/translator-resp-openai-responses.test.ts index 27cc877f2ef..2253c655ac7 100644 --- a/tests/unit/translator-resp-openai-responses.test.ts +++ b/tests/unit/translator-resp-openai-responses.test.ts @@ -423,6 +423,52 @@ test("Responses -> OpenAI: preserves non-object Read JSON-string arguments", () assert.equal(done.choices[0].delta.tool_calls[0].function.arguments, "null"); }); +test("Responses -> OpenAI: mixed plaintext + encrypted_content reasoning replays its plaintext (#10949)", () => { + const state = {}; + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { + type: "reasoning", + id: "rs_mixed", + content: [ + { + type: "reasoning_text", + text: "Let me start by reading the directory to understand the structure of the corpus.", + }, + ], + encrypted_content: "", + summary: [], + }, + }, + state + ); + + assert.ok(done, "mixed reasoning item must surface a delta"); + assert.equal( + done.choices[0].delta.reasoning_content, + "Let me start by reading the directory to understand the structure of the corpus." + ); +}); + +test("Responses -> OpenAI: opaque-only reasoning still emits no fabricated plaintext", () => { + const state = {}; + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { + type: "reasoning", + id: "rs_opaque_only", + encrypted_content: "", + summary: [], + }, + }, + state + ); + + assert.equal(done, null); +}); + test("Responses -> OpenAI: strips empty optional args from JSON-string output_item.done arguments", () => { const state = {}; openaiResponsesToOpenAIResponse( diff --git a/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx b/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx new file mode 100644 index 00000000000..b86f4188c6a --- /dev/null +++ b/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx @@ -0,0 +1,92 @@ +// @vitest-environment jsdom +/** + * CheaperInferenceSponsorBanner — render gate (localStorage dismissal), CTA + * pointing at our link.omniroute.online branded short link, and discreet + * partner-link note. Mirrors kimiSponsorBanner.test.tsx, minus the version gate + * (this banner is a durable partnership, not a time-boxed offer). + */ +import React from "react"; +import { act } from "react"; +import { createRoot } from "react-dom/client"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; + +const STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; +const SHORT_URL = "https://link.omniroute.online/cheaper"; + +vi.mock("next-intl", () => ({ useTranslations: () => (k: string) => k })); +vi.mock("@/shared/components/ProviderIcon", () => ({ default: () => null })); + +async function renderBanner(): Promise { + vi.resetModules(); + const { default: CheaperInferenceSponsorBanner } = + await import("../../../src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner"); + + const container = document.createElement("div"); + document.body.appendChild(container); + const root = createRoot(container); + act(() => { + root.render(); + }); + return container; +} + +describe("CheaperInferenceSponsorBanner", () => { + beforeEach(() => { + ( + globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean } + ).IS_REACT_ACT_ENVIRONMENT = true; + localStorage.removeItem(STORAGE_KEY); + }); + + afterEach(() => { + document.body.innerHTML = ""; + localStorage.removeItem(STORAGE_KEY); + }); + + it("renders with the CTA pointing at the branded short link", async () => { + const container = await renderBanner(); + expect(container.textContent).toContain("title"); + expect(container.textContent).toContain("cta"); + const link = container.querySelector("a[href]"); + expect(link).not.toBeNull(); + expect(link?.getAttribute("href")).toBe(SHORT_URL); + expect(link?.getAttribute("target")).toBe("_blank"); + expect(link?.getAttribute("rel")).toContain("noopener"); + }); + + it("shows the discreet partner-link note near the CTA", async () => { + const container = await renderBanner(); + expect(container.textContent).toContain("partnerLinkNote"); + const link = container.querySelector("a[href]"); + expect(link?.getAttribute("title")).toBe("partnerLinkNote"); + }); + + it("hides after dismissal and stays hidden on re-render", async () => { + const first = await renderBanner(); + const button = first.querySelector("button"); + expect(button).not.toBeNull(); + act(() => { + button?.click(); + }); + expect(localStorage.getItem(STORAGE_KEY)).toBe("true"); + expect(first.textContent).not.toContain("title"); + + // a fresh render (simulating a later visit) stays hidden + const second = await renderBanner(); + expect(second.textContent).not.toContain("title"); + }); + + it("re-renders visible again only after the key is cleared", async () => { + const first = await renderBanner(); + const button = first.querySelector("button"); + act(() => { + button?.click(); + }); + expect(localStorage.getItem(STORAGE_KEY)).toBe("true"); + + localStorage.removeItem(STORAGE_KEY); + const second = await renderBanner(); + expect(second.textContent).toContain("title"); + }); +}); diff --git a/tests/unit/ui/vscodeCopilotBanner.test.tsx b/tests/unit/ui/vscodeCopilotBanner.test.tsx index ed75ad4950a..fad50b2b266 100644 --- a/tests/unit/ui/vscodeCopilotBanner.test.tsx +++ b/tests/unit/ui/vscodeCopilotBanner.test.tsx @@ -11,7 +11,7 @@ import { createRoot } from "react-dom/client"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; const STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; vi.mock("next-intl", () => ({ useTranslations: () => (k: string) => k })); diff --git a/tests/unit/usage-command-json-format.test.ts b/tests/unit/usage-command-json-format.test.ts new file mode 100644 index 00000000000..9f76d6e1913 --- /dev/null +++ b/tests/unit/usage-command-json-format.test.ts @@ -0,0 +1,162 @@ +/** + * #8 (OmniCopilot) — the usage command answered `text/plain`, which a UI cannot + * parse safely. The structured form (`?format=json`) returns the same + * `ApiKeyUsageLimitStatus` + `UsageSnapshot` the text is rendered from. + * + * These tests pin the contract the extension depends on: JSON when asked, + * text by default, the 403 as a structured reason rather than a bare string, + * and an error body that never carries a stack trace. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { handleInternalUsageCommandHttpRequest } from "../../src/lib/usage/internalUsageCommand"; + +const NOW = Date.parse("2026-08-19T12:00:00.000Z"); + +const LIMIT_STATUS = { + enabled: true, + dailyLimitUsd: 5, + weeklyLimitUsd: 20, + dailySpentUsd: 1.25, + weeklySpentUsd: 8, + dailyWindowStartIso: "2026-08-19T03:00:00.000Z", + dailyResetAtIso: "2026-08-20T03:00:00.000Z", + weeklyWindowStartIso: "2026-08-16T03:00:00.000Z", + weeklyResetAtIso: "2026-08-23T03:00:00.000Z", + dailyExceeded: false, + weeklyExceeded: false, +}; + +function allowedDeps(overrides: Record = {}) { + return { + now: () => NOW, + isValidApiKey: async (apiKey: string) => apiKey === "sk-allowed", + getApiKeyMetadata: async () => ({ + id: "key-allowed", + name: "panel key", + allowUsageCommand: true, + usageLimitEnabled: true, + }), + getProviderConnections: async () => [ + { id: "conn-claude", provider: "claude", isActive: true }, + { id: "conn-codex", provider: "codex", isActive: true }, + ], + getAllProviderLimitsCache: () => ({ + "conn-claude": { + plan: "Claude Max", + quotas: { + weekly: { used: 25, total: 100, remaining: 75, resetAt: "2026-08-25T03:00:00.000Z" }, + }, + message: null, + fetchedAt: new Date(NOW).toISOString(), + }, + "conn-codex": { + plan: "Codex Pro", + quotas: { + weekly: { used: 9, total: 100, remaining: 91, resetAt: "2026-08-24T03:00:00.000Z" }, + }, + message: null, + fetchedAt: new Date(NOW).toISOString(), + }, + }), + getProviderConnectionById: async () => null, + getProviderLimitsCache: () => null, + getQuotaPolicy: async () => ({ defaultThresholdPercent: 0, providerWindowDefaults: {} }), + getApiKeyUsageLimitStatus: async () => LIMIT_STATUS, + ...overrides, + }; +} + +test("om-usage ?format=json returns the structured personal + provider quota", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + assert.match(response.headers.get("content-type") ?? "", /application\/json/); + const body = (await response.json()) as { + allowed: boolean; + personal: { dailySpentUsd: number } | null; + provider: { provider: string; connectionId: string } | null; + }; + assert.equal(body.allowed, true); + assert.equal(body.personal?.dailySpentUsd, 1.25); + assert.equal(body.provider?.provider, "claude"); + assert.equal(body.provider?.connectionId, "conn-claude"); +}); + +test("om-usage without ?format stays text/plain (the historical contract)", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + assert.match(response.headers.get("content-type") ?? "", /text\/plain/); + const text = await response.text(); + assert.match(text, /Personal quota/); + assert.match(text, /Provider quota/); +}); + +test("om-usage ?format=json returns every connection under providers[], not just the selected one", async () => { + // #11191 — a panel needs Codex + Claude side by side; the single `provider` + // pick is a terminal presentation choice, the collector had them all. + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + const body = (await response.json()) as { + allowed: boolean; + provider: { provider: string } | null; + providers: Array<{ provider: string }>; + }; + assert.equal(body.allowed, true); + const names = body.providers.map((s) => s.provider).sort(); + assert.deepEqual(names, ["claude", "codex"]); + // the single-pick field is still present and one of them + assert.ok(["claude", "codex"].includes(body.provider?.provider ?? "")); +}); + +test("om-usage ?format=json reports a disallowed key as structured allowed:false", async () => { + // A usage panel must tell "this key may not ask" apart from "no data yet", + // which a bare 403 text body cannot express. + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps({ + getApiKeyMetadata: async () => ({ id: "key-off", allowUsageCommand: false }), + }) + ); + + assert.equal(response.status, 403); + const body = (await response.json()) as { allowed: boolean }; + assert.equal(body.allowed, false); +}); + +test("om-usage ?format=json rejects an invalid key and never leaks a stack trace", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-wrong" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 401); + const body = (await response.json()) as { allowed: boolean; error?: { message?: string } }; + assert.equal(body.allowed, false); + assert.ok( + !body.error?.message?.includes("at /"), + "error bodies must not carry stack frames (ERROR_SANITIZATION)" + ); +}); diff --git a/tests/unit/validate-response-quality.test.ts b/tests/unit/validate-response-quality.test.ts index de4b5f47dc3..6a918155e56 100644 --- a/tests/unit/validate-response-quality.test.ts +++ b/tests/unit/validate-response-quality.test.ts @@ -24,6 +24,20 @@ test("returns valid=true for SSE with 'data:' lines", async () => { assert.strictEqual(res.valid, true); }); +test("returns valid=true for SSE opening with a ':' comment line (e.g. OpenRouter keep-alive)", async () => { + const res = await validateResponseQuality( + makeResponse(': OPENROUTER PROCESSING\n\ndata: {"foo":"bar"}\n\n'), + false, + {} + ); + assert.strictEqual(res.valid, true); +}); + +test("returns valid=true for an SSE stream with leading whitespace before the first frame", async () => { + const res = await validateResponseQuality(makeResponse('\n\ndata: {"foo":"bar"}\n\n'), false, {}); + assert.strictEqual(res.valid, true); +}); + test("returns valid=false for non-JSON non-SSE text", async () => { const res = await validateResponseQuality(makeResponse("Hello world"), false, {}); assert.strictEqual(res.valid, false); diff --git a/tests/unit/vision-bridge-maxchars.test.ts b/tests/unit/vision-bridge-maxchars.test.ts index d667e515621..7f364468c73 100644 --- a/tests/unit/vision-bridge-maxchars.test.ts +++ b/tests/unit/vision-bridge-maxchars.test.ts @@ -67,6 +67,11 @@ test("modalityBridgeVisionMaxChars=120 caps the description with a … suffix", }), callVisionModel: async (_imageDataUri: string, _config: VisionModelConfig) => LONG_DESCRIPTION, + // #10859 made the reroute heuristic try a live vision-capable model for + // not-combo text-only models, which would hijack the request before the + // describe path. Pin credentials to definitively-unusable (false) so the + // reroute is excluded and the describe path under test runs. + hasUsableCredentials: async () => false, }, }); @@ -93,6 +98,8 @@ test("no modalityBridgeVisionMaxChars key: description is passed through in full }), callVisionModel: async (_imageDataUri: string, _config: VisionModelConfig) => LONG_DESCRIPTION, + // Same #10859 reroute guard as above — keep the describe path under test. + hasUsableCredentials: async () => false, }, }); @@ -125,6 +132,8 @@ test("updateSettingsSchema accepts an explicit modalityBridgeVisionMaxChars: 0 t }), callVisionModel: async (_imageDataUri: string, _config: VisionModelConfig) => LONG_DESCRIPTION, + // Same #10859 reroute guard as above — keep the describe path under test. + hasUsableCredentials: async () => false, }, }); diff --git a/tests/unit/zcode-executor.test.ts b/tests/unit/zcode-executor.test.ts index c82e7b33683..db4a8586bee 100644 --- a/tests/unit/zcode-executor.test.ts +++ b/tests/unit/zcode-executor.test.ts @@ -26,6 +26,8 @@ function requestBody() { test("ZCode accepts GLM Coding Plan models and rejects unsafe/unknown ids", async () => { const { resolveZcodeModel } = await loadZcodeExecutor(); assert.deepEqual(resolveZcodeModel("glm-5.2"), { ok: true, model: "glm-5.2" }); + assert.equal(resolveZcodeModel("glm-5.2-high").ok, false); + assert.equal(resolveZcodeModel("glm-5.3-low").ok, false); assert.equal(resolveZcodeModel("-unexpected").ok, false); assert.equal(resolveZcodeModel("unknown-model").ok, false); }); @@ -70,7 +72,7 @@ test("ZCode buffers the completed turn into OpenAI SSE when stream=true", async }); const result = await executor.execute({ - model: "glm-5.2-high", + model: "glm-5.2", body: requestBody(), stream: true, credentials: {}, diff --git a/tests/unit/zcode-provider.test.ts b/tests/unit/zcode-provider.test.ts index 3acf3c862c4..e882ab5fd91 100644 --- a/tests/unit/zcode-provider.test.ts +++ b/tests/unit/zcode-provider.test.ts @@ -10,5 +10,18 @@ test("ZCode provider registry exposes a local no-auth GLM Coding Plan backend", assert.equal(zcodeProvider.baseUrl, "zcode://app-server/stdio"); assert.equal(zcodeProvider.authType, "none"); assert.equal(zcodeProvider.authHeader, "none"); - assert.equal(zcodeProvider.models.some((model) => model.id === "glm-5.2"), true); + assert.equal( + zcodeProvider.models.some((model) => model.id === "glm-5.2"), + true + ); + for (const alias of ["glm-5.3-high", "glm-5.3-low", "glm-5.2-high", "glm-5.2-max"]) { + assert.equal( + zcodeProvider.models.some((model) => model.id === alias), + false, + alias + ); + } + for (const model of zcodeProvider.models) { + assert.deepEqual(model.supportedThinkingEfforts, [], model.id); + } });