diff --git a/.env.example b/.env.example index 9616acd3189e..84bde3f0256f 100644 --- a/.env.example +++ b/.env.example @@ -87,6 +87,10 @@ DISABLE_SQLITE_AUTO_BACKUP=false # Used by: src/shared/utils/rateLimiter.ts # Example: redis://localhost:6379 (or redis://redis:6379 in Docker) # REDIS_URL=redis://localhost:6379 +# Namespace prefix for ALL OmniRoute Redis keys (rate limiter + auth cache + +# quota store). Prevents key collisions when OmniRoute shares a Redis instance +# with other apps (e.g. on 127.0.0.1:6379). Default when unset: omniroute: +# REDIS_KEY_PREFIX=omniroute: # Host interface docker-compose publishes the Redis sidecar on. # Default: 127.0.0.1 (loopback only). The compose Redis runs WITHOUT # `requirepass`, and app containers reach it over the compose network @@ -372,9 +376,8 @@ ALLOW_API_KEY_REVEAL=false # NO_LOG_API_KEY_IDS=key_abc123,key_def456 # Fallback per-day request budget applied to API keys whose `rate_limits` -# column is null. Default (unset/empty/malformed) preserves the legacy -# 1000/day, 5000/week, 20000/month windows so existing deployments do not -# silently lose rate limiting on upgrade. +# column is null. Default (unset/empty) is unlimited (no implicit caps). +# Malformed values preserve the legacy 1000/day, 5000/week, 20000/month windows. # Set explicitly to "0" to opt out entirely (unlimited fallback). Any # positive integer N enables N/day, 5N/week, 20N/month. # Used by: src/shared/utils/apiKeyPolicy.ts — checkRateLimit() fallback. diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index d8a65576dc67..3b04a20c8bb5 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -183,15 +183,55 @@ jobs: env: DOCKER_BUILDKIT_INLINE_CACHE: 1 + - name: Build and push BUN base platform image by digest + id: build-bun-base + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 + with: + context: . + file: Dockerfile.bun + target: runner-base + platforms: ${{ matrix.platform }} + outputs: type=image,push-by-digest=true,name-canonical=true,push=true + tags: | + ${{ env.IMAGE_NAME }} + ${{ env.GHCR_IMAGE_NAME }} + cache-from: type=gha,scope=docker-bun-base-${{ matrix.arch }} + cache-to: type=gha,scope=docker-bun-base-${{ matrix.arch }},mode=max + no-cache: false + env: + DOCKER_BUILDKIT_INLINE_CACHE: 1 + + - name: Build and push BUN web platform image by digest + id: build-bun-web + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 + with: + context: . + file: Dockerfile.bun + target: runner-web + platforms: ${{ matrix.platform }} + outputs: type=image,push-by-digest=true,name-canonical=true,push=true + tags: | + ${{ env.IMAGE_NAME }} + ${{ env.GHCR_IMAGE_NAME }} + cache-from: type=gha,scope=docker-bun-web-${{ matrix.arch }} + cache-to: type=gha,scope=docker-bun-web-${{ matrix.arch }},mode=max + no-cache: false + env: + DOCKER_BUILDKIT_INLINE_CACHE: 1 + - name: Export digests env: DIGEST_BASE: ${{ steps.build.outputs.digest }} DIGEST_WEB: ${{ steps.build-web.outputs.digest }} + DIGEST_BUN_BASE: ${{ steps.build-bun-base.outputs.digest }} + DIGEST_BUN_WEB: ${{ steps.build-bun-web.outputs.digest }} run: | set -euo pipefail - mkdir -p /tmp/digests/base /tmp/digests/web + mkdir -p /tmp/digests/base /tmp/digests/web /tmp/digests/bun-base /tmp/digests/bun-web touch "/tmp/digests/base/${DIGEST_BASE#sha256:}" touch "/tmp/digests/web/${DIGEST_WEB#sha256:}" + touch "/tmp/digests/bun-base/${DIGEST_BUN_BASE#sha256:}" + touch "/tmp/digests/bun-web/${DIGEST_BUN_WEB#sha256:}" - name: Upload base digests uses: actions/upload-artifact@v7 @@ -209,6 +249,22 @@ jobs: if-no-files-found: error retention-days: 1 + - name: Upload bun-base digests + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: digests-bun-base-${{ matrix.arch }} + path: /tmp/digests/bun-base/* + if-no-files-found: error + retention-days: 1 + + - name: Upload bun-web digests + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: digests-bun-web-${{ matrix.arch }} + path: /tmp/digests/bun-web/* + if-no-files-found: error + retention-days: 1 + merge: name: Publish multi-arch manifests needs: @@ -263,6 +319,20 @@ jobs: path: /tmp/digests/web merge-multiple: true + - name: Download bun-base digests + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: digests-bun-base-* + path: /tmp/digests/bun-base + merge-multiple: true + + - name: Download bun-web digests + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: digests-bun-web-* + path: /tmp/digests/bun-web + merge-multiple: true + - name: Create Docker Hub manifest run: | set -euo pipefail @@ -286,6 +356,8 @@ jobs: create_manifest "${IMAGE_NAME}" "" /tmp/digests/base create_manifest "${IMAGE_NAME}" "-web" /tmp/digests/web + create_manifest "${IMAGE_NAME}" "-bun" /tmp/digests/bun-base + create_manifest "${IMAGE_NAME}" "-web-bun" /tmp/digests/bun-web - name: Create GHCR manifest run: | @@ -310,6 +382,8 @@ jobs: create_manifest "${GHCR_IMAGE_NAME}" "" /tmp/digests/base create_manifest "${GHCR_IMAGE_NAME}" "-web" /tmp/digests/web + create_manifest "${GHCR_IMAGE_NAME}" "-bun" /tmp/digests/bun-base + create_manifest "${GHCR_IMAGE_NAME}" "-web-bun" /tmp/digests/bun-web - name: Inspect image if: needs.prepare.outputs.version != 'main' diff --git a/AGENTS.md b/AGENTS.md index 090bcf41cd63..d468f7d19840 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -46,7 +46,7 @@ Repository map and Reference Documentation sections below. ## Project at a Glance -**OmniRoute** — unified AI proxy/router. One endpoint, 348 LLM providers, auto-fallback. +**OmniRoute** — unified AI proxy/router. One endpoint, 349 LLM providers, auto-fallback. | Layer | Location | Purpose | | ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | @@ -56,7 +56,7 @@ Repository map and Reference Documentation sections below. | Translators | `open-sse/translator/` | Format conversion (OpenAI↔Claude↔Gemini) | | Transformer | `open-sse/transformer/` | Responses API ↔ Chat Completions | | Services | `open-sse/services/` | Combo routing, rate limits, caching, etc | -| Database | `src/lib/db/` | SQLite domain modules (157 migrations) | +| Database | `src/lib/db/` | SQLite domain modules (159 migrations) | | Domain/Policy | `src/domain/` | Policy engine, cost rules, fallback logic | | MCP Server | `open-sse/mcp-server/` | 110 tools (44 canonical + memory/skill/GitHub/pool/gamification/plugin/Notion/Obsidian/local-corpus/RTK modules), 3 transports (stdio / SSE / Streamable HTTP), 33 scopes | | A2A Server | `src/lib/a2a/` | JSON-RPC 2.0 agent protocol | diff --git a/Dockerfile.bun b/Dockerfile.bun index 1804d3f78853..bb547ce210da 100644 --- a/Dockerfile.bun +++ b/Dockerfile.bun @@ -49,8 +49,8 @@ ENV NODE_ENV=production # Bun native Next.js build execution RUN bun run --quiet build -# ── Runner stage (100% Bun Native Production Runtime) ────────────────────── -FROM oven/bun:1.3.14-slim AS runner +# ── Runner Base stage (100% Bun Native Production Runtime) ────────────────── +FROM oven/bun:1.3.14-slim AS runner-base LABEL org.opencontainers.image.title="omniroute" \ org.opencontainers.image.description="Unified AI proxy — route any LLM through one endpoint (Bun Native)" \ @@ -86,4 +86,61 @@ EXPOSE 20128 HEALTHCHECK --interval=30s --timeout=5s --start-period=15s --retries=3 \ CMD bun healthcheck.mjs || exit 1 -ENTRYPOINT ["bun", "bin/omniroute.mjs", "serve", "--no-open"] +ENTRYPOINT ["bun", "dev/run-standalone.mjs"] + +# ── Runner Web stage (Bun Native + Chromium/Playwright for Web providers) ─── +FROM runner-base AS runner-web + +USER root + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + chromium \ + chromium-driver \ + fonts-liberation \ + libasound2t64 \ + gconf-service \ + libatk-bridge2.0-0 \ + libatk1.0-0 \ + libc6 \ + libcairo2 \ + libcups2 \ + libdbus-1-3 \ + libexpat1 \ + libfontconfig1 \ + libgbm1 \ + libgcc-s1 \ + libglib2.0-0 \ + libgtk-3-0 \ + libnspr4 \ + libnss3 \ + libpango-1.0-0 \ + pangocairo-1.0-0 \ + stdc++6 \ + libx11-6 \ + libx11-xcb1 \ + libxcb1 \ + libxcomposite1 \ + libxcursor1 \ + libxdamage1 \ + libxext6 \ + libxfixes3 \ + libxi6 \ + libxrandr2 \ + libxrender1 \ + libxss1 \ + libxtst6 \ + ca-certificates \ + fonts-gargi \ + fonts-ipafont-gothic \ + fonts-kacst \ + fonts-thai-tlwg \ + fonts-wqy-zenhei \ + && rm -rf /var/lib/apt/lists/* + +ENV PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD=1 +ENV PLAYWRIGHT_CHROMIUM_EXECUTABLE_PATH=/usr/bin/chromium + +# Return to the base image non-root user after the apt install (mirrors the +# Node Dockerfile runner-web stage, which re-asserts USER node). +USER bun diff --git a/PROVIDER_REFERENCE.md b/PROVIDER_REFERENCE.md new file mode 100644 index 000000000000..571fe0e904cf --- /dev/null +++ b/PROVIDER_REFERENCE.md @@ -0,0 +1,447 @@ +--- +title: "Provider Reference" +version: 3.8.50 +lastUpdated: 2026-08-21 +--- + +# Provider Reference + +> **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. +> Regenerate with: `npm run gen:provider-reference` +> **Last generated:** 2026-08-21 + +Total providers: **349**. See category breakdown below. + +## Categories + +- **Free** — free tier with API key (configured via dashboard) +- **No-auth** — public endpoints that require no key or sign-in at all +- **OAuth** — sign-in flow handled by OmniRoute, no API key needed +- **Web cookie** — wraps the provider's web app via cookie auth +- **API key** — paid provider configured via API key (free credits may apply) +- **Local** — runs on the user's machine (Ollama, LM Studio, vLLM, etc.) +- **Search** — web search providers +- **Audio** — audio-only providers (TTS/STT) +- **Upstream proxy** — providers that proxy to other providers +- **Cloud agent** — long-running coding agents (Codex Cloud, Devin, Jules) +- **System** — OmniRoute-internal providers (loopback, etc.) + +Additional tags: `image`, `video`, `aggregator`, `enterprise`, `embed/rerank`, `self-hosted`. + +`Tool calling` (where shown): `native` — real function-calling API; `emulated` — the `tools` array is prompt-emulated via `webTools.ts` (regex-parsed `{...}` blocks); `none` — `tools` is currently silently dropped. See #7286. + +Use the dashboard at `/dashboard/providers` to enable, configure, and test each provider. + +--- + +## No-auth Providers (no key required) (11) + +| ID | Alias | Name | Tags | Website | Notes | Tool calling | +|----|-------|------|------|---------|-------|--------------| +| `aihorde` | `horde` | AI Horde | No-auth | [link](https://aihorde.net) | No API key required — uses AI Horde's documented anonymous key. Adding a free aihorde.net key is optional and only buys higher queue priority (kudos). | — | +| `auggie` | `aug` | Augment (Auggie CLI) | No-auth | [link](https://augmentcode.com) | No API key stored by OmniRoute. Install the Auggie CLI and run `auggie login` on this machine, then OmniRoute spawns it locally for each request. | — | +| `chipotle` | `pepper` | Chipotle Pepper AI (Free) | No-auth | [link](https://amelia.chipotle.com) | No credentials required. Uses Chipotle's public support chatbot via reverse-engineered SockJS/STOMP protocol. | — | +| `cloudflare-playground` | `cfp` | Cloudflare AI Playground | No-auth | [link](https://playground.ai.cloudflare.com) | No credentials required — anonymous browser sessions over a reverse-engineered cf_agent WebSocket protocol (Playwright transport). | — | +| `devin-cli-agentic` | `dva` | Devin CLI Agentic Bridge | No-auth | [link](https://docs.devin.ai/work-with-devin/devin-cli) | Authentication is owned by the official Devin CLI in its isolated bridge volume. | emulated | +| `duckduckgo-web` | `ddgw` | DuckDuckGo AI Chat | No-auth | [link](https://duckduckgo.com/duckchat) | No credentials required — DuckDuckGo AI Chat is anonymous and free. | emulated | +| `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — | +| `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — | +| `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — | +| `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — | +| `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — | + +## OAuth Providers (25) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | +| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | +| `antigravity` | — | Antigravity | OAuth | — | — | +| `claude` | `cc` | Claude Code | OAuth | — | — | +| `cline` | `cl` | Cline | OAuth | — | — | +| `clinepass` | `cp` | ClinePass | OAuth | [link](https://cline.bot/cline-pass) | ClinePass is Cline's $9.99/mo subscription bundling 10 open coding models. Sign in with your Cline account (same login as the Cline CLI/IDE), or paste a direct ClinePass API key (app.cline.bot → Settings → API Keys). A ClinePass subscription unlocks the cline-pass/* models. Reuses the Cline WorkOS OAuth flow. | +| `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. | +| `codex` | `cx` | OpenAI Codex | OAuth | — | — | +| `cursor` | `cu` | Cursor IDE | OAuth | — | — | +| `devin-cli` | `dv` | Devin CLI | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | +| `devin-desktop` | — | Devin Desktop | OAuth | [link](https://devin.ai) | Paste an existing Devin API key from an authenticated Devin session. Key export availability and steps vary by Devin version and account. | +| `ghe-copilot` | `ghe-copilot` | GitHub Enterprise Copilot | OAuth | — | Enter your GHE instance URL (e.g., https://ghe.company.com) in provider settings, then authenticate via device flow. | +| `github` | `gh` | GitHub Copilot | OAuth | — | — | +| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab Duo OAuth is not configured. Register an OAuth application at https://gitlab.com/-/profile/applications with redirect URI http://localhost:20128/callback and scopes "ai_features read_user", then set GITLAB_DUO_OAUTH_CLIENT_ID (and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET) and restart. | +| `grok-cli` | `gc` | Grok Build | OAuth | — | Sign in with your browser, or paste your ~/.grok/auth.json (or the JWT access token) from the Grok Build CLI; refresh_token is rotated automatically either way. | +| `kilocode` | `kc` | Kilo Code | OAuth | — | — | +| `kimi-coding` | `kmc` | Kimi Code CLI | OAuth | [link](https://www.kimi.com/code?aff=omniroute) | Sign in with the same Kimi account used by Kimi Code CLI. OmniRoute uses the CLI OAuth flow and Kimi Coding Plan endpoints. | +| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | +| `openference` | `of` | Openference | OAuth | [link](https://openference.com) | Sign in with your Openference account to route requests through api.openference.com. An active plan is required for inference — OAuth may authenticate but return 402 without one. | +| `qoder` | `if` | Qoder | OAuth | — | — | +| `raycast` | `rc` | Raycast Pro AI | OAuth | [link](https://raycast.com/ai) | Unofficial integration — uses your Raycast Pro subscription via credentials from the macOS app (Auto-Import or manual capture). May break on Raycast updates. Not for redistribution; personal use only. | +| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | +| `xai-oauth` | `xao` | xAI OAuth (Grok) | OAuth | [link](https://x.ai) | Sign in with xAI to use api.x.ai models such as Grok 4.5. This is separate from Grok Build JWT sessions, which use cli-chat-proxy.grok.com and grok-build model aliases. | +| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | +| `zed-hosted` | — | Zed Hosted Models | OAuth | [link](https://zed.dev) | Sign in with your Zed account (native-app sign-in). OmniRoute generates a one-time RSA keypair and opens zed.dev to authorize it — on a remote/headless install, copy the resulting 127.0.0.1 callback URL from your browser's address bar and paste it back here. Distinct from the 'Zed IDE' credential-import entry above: this proxies chat completions through Zed's own hosted model aggregator (cloud.zed.dev), fronting Anthropic/OpenAI/Google/xAI models under your Zed plan. | + +## Web Cookie Providers (35) + +| ID | Alias | Name | Tags | Website | Notes | Tool calling | +|----|-------|------|------|---------|-------|--------------| +| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | emulated | +| `adobe-firefly` | `firefly` | Adobe Firefly (Image/Video) | Web cookie | [link](https://firefly.adobe.com) | RECOMMENDED: firefly.adobe.com signed-in → F12 → Network → click firefly-3p.ff.adobe.io (generate-async or models/discovery) → Request Headers → Authorization → copy the token AFTER 'Bearer ' (starts with eyJ…). Cookie-only from firefly.adobe.com mints a GUEST token → 401/403; only multi-domain IMS cookies (adobelogin.com) or that Bearer JWT work. Unofficial/experimental media + Limits. | — | +| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | emulated | +| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | emulated | +| `chatgpt-web-codex` | `cgpt-codex` | ChatGPT Web (Codex) | Web cookie | [link](https://chatgpt.com) | Paste the full ChatGPT Cookie header. OmniRoute verifies it in an isolated headless browser profile. | native | +| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | none | +| `conol-web` | `cnl` | Conol (Unofficial/Experimental) | Web cookie | [link](https://conol.ai) | Use browser sign-in, or paste the full Cookie header from conol.ai. The __Secure-better-auth.session_token cookie is required. | — | +| `copilot-m365-web` | `m365copilot` | Microsoft 365 Copilot (BizChat) | Web cookie | [link](https://m365.cloud.microsoft/chat) | Sign in at m365.cloud.microsoft/chat, then open DevTools → Network → filter 'WS' → click the Chathub WebSocket connection. Copy both the access_token query parameter AND the account-specific Chathub path segment from its request URL (wss://…/Chathub/?…&access_token=…). It is NOT an Authorization: Bearer header on an XHR/Fetch request. The token is short-lived; this is an unofficial integration. Optional: store a refresh_token in providerSpecificData.refreshToken (any Microsoft device-code/refresh flow for the substrate.office.com/sydney scopes) and OmniRoute pre-flight-refreshes the access token itself — otherwise re-capture after every ~75 min expiry. | — | +| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste the access_token from an authenticated copilot.microsoft.com request (DevTools → Network → Authorization), or export a HAR while logged in | — | +| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | emulated | +| `doubao-web` | `db` | Dola Web (ByteDance) | Web cookie | [link](https://www.dola.com) | Paste the full Cookie header from www.dola.com. It should include sessionid, ttwid, and s_v_web_id. If s_v_web_id is unavailable, fp=verify_... from a chat/completion request URL can be used as a fallback. | — | +| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — | +| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated | +| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — | +| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | +| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — | +| `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — | +| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated | +| `kimi-web` | `kimi-web` | Kimi Web | Web cookie | [link](https://www.kimi.com/code?aff=omniroute) | Paste access_token from www.kimi.com DevTools → Application → Local Storage. A legacy kimi-auth cookie is also accepted. | — | +| `lmarena` | `lma` | Arena (Free) | Web cookie | [link](https://arena.ai) | Paste the full Cookie header from arena.ai (DevTools → Network → request → Cookie). Include arena-auth-prod-v1.0/.1… and cf_clearance/__cf_bm when present. OmniRoute uses Chrome TLS impersonation; if Arena still 403s, set providerSpecificData.recaptchaV3Token from a live browser session. | — | +| `microsoft-designer-web` | `msdesigner` | Microsoft Designer (Image Generation) | Web cookie | [link](https://designer.microsoft.com) | Sign in at designer.microsoft.com, then open DevTools → Network, generate an image, and find the request to DallE.ashx?action=GetDallEImagesCogSci. Copy the value of its Authorization: Bearer header (the access_token — no 'Bearer ' prefix). The token is short-lived; this is an unofficial, reverse-engineered integration. | — | +| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your ecto_1_sess cookie AND the ecto1:... WS auth token from meta.ai. Capture the ecto1: token in DevTools → Network → WS → the clippy request's Authorization query param. Example: ecto_1_sess=4240a308...NVDg0; ecto1:ABCD... | emulated | +| `notion-web` | `nw` | Notion AI Web (Unofficial/Experimental) | Web cookie | [link](https://www.notion.so) | Paste only the token_v2 cookie VALUE from app.notion.com (DevTools → Application → Cookies → token_v2). Do not paste token_v2= or the full Cookie header. Workspace is auto-detected; space_id / notion_user_id are optional. | — | +| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | emulated | +| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | — | +| `promptql` | `pql` | PromptQL (Unofficial/Experimental) | Web cookie | [link](https://prompt.ql.app) | Paste the Bearer JWT from prompt.ql.app DevTools → Network → graphql → Authorization (token only). Optional projectId + session Cookie for refresh. | — | +| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | emulated | +| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | emulated | +| `tencent-aistudio-web` | `tasw` | Tencent AI Studio (Free) | Web cookie | [link](https://aistudio.tencent.ai) | Log in to aistudio.tencent.ai, open DevTools -> Network, copy any request Cookie header containing session tokens. | — | +| `tinycms-web` | `tcw` | TinyCMS Web (Free/Sub) | Web cookie | [link](https://site.tinycms.xyz) | Go to site.tinycms.xyz, open DevTools → Application → Local Storage, copy the value of 'app-config-uuid' (starts with 'R'), and paste it here. | — | +| `v0-vercel-web` | `v0-vercel-web` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | — | +| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | — | +| `yuanbao-web` | `ybw` | Tencent Yuanbao (Free) | Web cookie | [link](https://yuanbao.tencent.com) | Log in to yuanbao.tencent.com, then paste the full Cookie header (DevTools → Network → any /api request → Request Headers → Cookie). It must contain hy_user and hy_token. | — | +| `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | +| `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | + +## API Key Providers (paid / paid-with-free-credits) (233) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | +| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | +| `agnes` | `agnes` | Agnes AI | API key, video | [link](https://agnes-ai.com) | Get API key at agnes-ai.com | +| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | +| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | Free tier paused (2026) — AI/ML API is now pay-as-you-go only (min $20 top-up); no recurring free credits. | +| `ainative` | `ainative` | AINative Studio | API key | [link](https://ainative.studio) | Create a free API key at ainative.studio (no card), then paste it here as a Bearer token. | +| `aion` | `aion` | Aion Labs | API key | [link](https://www.aionlabs.ai) | Create a free API key at aionlabs.ai (no card), then paste it here as a Bearer token. | +| `alibaba` | `ali` | Alibaba Cloud Model Studio | API key | [link](https://bailian.console.alibabacloud.com/) | — | +| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.console.aliyun.com/) | — | +| `ant-ling` | `ling` | Ant Ling / Ring (inclusionAI) | API key | [link](https://developer.ant-ling.com/en/docs/) | Register and create an API key at the Ant Ling API console (https://chat.ant-ling.com/open), then paste it here. OmniRoute routes chat traffic to https://api.ant-ling.com/v1/chat/completions; the provider is OpenAI-compatible and also exposes an Anthropic-compatible surface. | +| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | +| `anyapi` | `anyapi` | AnyAPI AI | API key, aggregator | [link](https://anyapi.ai) | Free plan: 100,000 ANY Tokens/day and 100 RPM for eligible Free/Basic models; no credit card required. | +| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | +| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | +| `auriko` | `auriko` | Auriko | API key, aggregator | [link](https://www.auriko.ai) | Free plan publishes 1,000 Platform RPM and 10,000 BYOK RPM. Platform inference still passes through provider cost; this is not a free-token pool or unlimited free inference. | +| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | +| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | +| `bai` | `bai` | b.ai | API key | [link](https://b.ai) | Bearer API key for the b.ai OpenAI-compatible LLM gateway (distinct from TheB.AI). Create a key at https://docs.b.ai, then use https://api.b.ai/v1 as the OpenAI-compatible base URL. | +| `baichuan` | `baichuan` | Baichuan | API key | [link](https://www.baichuan-ai.com/) | Get API key at platform.baichuan-ai.com | +| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://ernie.baidu.com/) | Get API key at console.bce.baidu.com | +| `bailian-coding-plan` | `bcp` | Alibaba Token Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/token-plan-overview) | — | +| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | +| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | +| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | +| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply | +| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | +| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | +| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | +| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | +| `charm-hyper` | `charm-hyper` | Charm Hyper | API key | [link](https://hyper.charm.land) | 100 free monthly Hypercredits on signup | +| `chat-oripe` | `chat-oripe` | Chat Oripe | API key, aggregator | [link](https://api.oriper.com) | Official metadata advertises 2M tokens/month, but the public site and documentation were blocked during audit; treat the quota and brand mapping as unconfirmed. | +| `chatanywhere` | `chatanywhere` | ChatAnywhere | API key, aggregator | [link](https://chatanywhere.tech) | Personal, educational or research use only: public documentation cites 10,000 points/day and 200 requests/day per IP/key; do not use for commercial traffic. | +| `cheaperinference` | `cinf` | Cheaper Inference | API key | [link](https://cheaperinference.com/?utm_source=omniroute) | — | +| `chenzk` | `chenzk` | Chenzk API | API key | [link](https://chenzk.top) | — | +| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | +| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | +| `cloudcode-one` | `cloudcode-one` | CloudCode.ONE | API key, aggregator | [link](https://cloudcode.one) | Published free models include glm-4.7-flash and glm-4.6v-flash; no numeric quota is published, and key creation may require credit or a coupon. | +| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | +| `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — | +| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | +| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | +| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | +| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | +| `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. | +| `dahl` | `dahl` | Dahl | API key | [link](https://inference.dahl.global) | Click 'Add Account' to auto-generate a token, or add a manual API key. | +| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | +| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | +| `deepai` | `deepai` | DeepAI | API key, image | [link](https://deepai.org) | Use your DeepAI API key. Get one at deepai.org — requires a Pro subscription ($9.99/mo). | +| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | +| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | +| `dgrid` | `dgrid` | DGrid | API key | [link](https://dgrid.ai) | DGrid Free Models Router: 10 requests/minute and 100 requests/day. A $5 lifetime top-up unlocks up to 20 requests/minute and 1,000 requests/day. | +| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | +| `digitalocean` | `digitalocean` | DigitalOcean | API key | [link](https://docs.digitalocean.com/products/ai-platform/) | — | +| `dit` | `dai` | DIT.ai | API key | [link](https://dit.ai) | Use your dit.ai API key in Authorization: Bearer . Fully OpenAI-compatible — a drop-in replacement, just change the base URL to https://api.dit.ai/v1. | +| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | +| `dxnt` | `dxnt` | DXNT / DX Token | API key, aggregator | [link](https://www.dxnt.com) | Free accounts are documented at 100 calls/day; the quota may increase through invitations and can vary by account. | +| `electronhub` | `electronhub` | Electron Hub | API key, aggregator | [link](https://www.electronhub.ai) | Free plan: 5 RPM, $0.25 weekly credits and 10 Neutrinos/day for :free models; family budgets also apply. | +| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | +| `factory` | `factory` | Factory | API key | [link](https://factory.ai) | Bearer API key for the Factory OpenAI-compatible gateway. | +| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | +| `fastrouter` | `fastrouter` | FastRouter | API key, aggregator | [link](https://fastrouter.ai) | Models with the :free suffix allow 10 requests/day per organization and model; availability may change. | +| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | +| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | +| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | +| `free-ai` | `free-ai` | Free.ai | API key, aggregator | [link](https://free.ai) | 30,000 tokens/day cover self-hosted models after email verification. Usage beyond the pool can bill at raw cost, and premium external models are paid. | +| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | +| `freebuff` | `freebuff` | Freebuff | API key | [link](https://freebuff.com) | Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester). | +| `freeinference` | `freeinference` | FreeInference | API key, aggregator | [link](https://freeinference.org) | Free research access without a card; non-Harvard applicants require manual approval and no numeric quota is publicly guaranteed. | +| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | +| `freetheai` | `fta` | FreeTheAi | API key, aggregator | [link](https://freetheai.xyz) | Join the FreeTheAi Discord to get your free API key. | +| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | +| `g4f-gemini` | `g4fgem` | g4f.space — Gemini | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-groq` | `g4fgroq` | g4f.space — Groq | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-nvidia` | `g4fnv` | g4f.space — NVIDIA | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-ollama` | `g4foll` | g4f.space — Ollama | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-pollinations` | `g4fpol` | g4f.space — Pollinations | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | ⚠️ **DEPRECATED.** api.galadriel.ai no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | +| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free tier available through Google AI Studio; current per-model quotas and regional limits apply | +| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | +| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | +| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | +| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free MiMo (xiaomi/mimo-v2.5) revoked 2026-05 — Opengateway is now a pay-as-you-go credit gateway; no recurring free model. | +| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free Nemotron promo ended 2026-06 — the GMI Cloud route is now pay-as-you-go credit only. | +| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | +| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | +| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | +| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | +| `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | +| `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | +| `helyxai` | `helyxai` | Helyx AI | API key, aggregator | [link](https://helyxai.space) | Operational Free plan documents 100,000 tokens/day; the site's separate 2M+ marketing claim conflicts and is not treated as a quota guarantee. | +| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | +| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | +| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | +| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | +| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `inception` | `inception` | Inception | API key | [link](https://docs.inceptionlabs.ai) | 10M free tokens on signup, no credit card required. | +| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | +| `internlm` | `internlm` | InternLM (Intern-S1) | API key | [link](https://internlm.intern-ai.org.cn/) | Free monthly quota ~1M input / 3M output tokens (~10 RPM) | +| `jina-ai` | `jina` | Jina AI (Foundation API) | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for api.jina.ai — embeddings, rerank, classify, segment, and search. Dashboard keys take precedence over JINA_AI_API_KEY. This is not the Reader / r.jina.ai card and does not fetch URLs. | +| `jina-reader` | `jr` | Jina Reader (r.jina.ai) | API key | [link](https://jina.ai/reader) | Bearer API key for r.jina.ai URL-to-markdown (/v1/web/fetch only). Does not serve /v1/embeddings or /v1/rerank. The same Jina token as Foundation API works; OmniRoute reuses a jina-ai dashboard key or JINA_AI_API_KEY when this card is empty. | +| `kenari` | `kenari` | Kenari | API key | [link](https://kenari.id) | Use your Kenari API key (kn-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://kenari.id/v1. | +| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | +| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | +| `kimi` | `kimi` | Kimi (Legacy Moonshot API) | API key | [link](https://platform.kimi.ai?aff=omniroute) | — | +| `kimi-coding-apikey` | `kmca` | Kimi Code API Key | API key | [link](https://www.kimi.com/code?aff=omniroute) | — | +| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | +| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | +| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | +| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | +| `literouter` | `literouter` | LiteRouter | API key, aggregator | [link](https://literouter.com) | Free model variants use the :free suffix; daily credit limits vary by model and free input is capped at 5,000 tokens. | +| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | +| `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. | +| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | +| `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | +| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. | +| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | +| `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | +| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | +| `meganova-ai` | `meganova-ai` | MegaNova AI | API key, aggregator | [link](https://meganova.ai) | Free signup without a card. Published Tier 1 per-model quotas total 550 requests/day; they are not a shared global pool, and paid overage can apply if enabled. | +| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | +| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | +| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | +| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | +| `mixedbread` | `mxbai` | Mixedbread AI | API key | [link](https://www.mixedbread.com) | Bearer API key for the Mixedbread embeddings API. | +| `mixlayer` | `mixlayer` | Mixlayer | API key, aggregator | [link](https://www.mixlayer.com) | The qwen/qwen3.5-4b-free model is free for prototyping and rate-limited; no fixed public RPM or daily quota is confirmed. | +| `mnn-ai` | `mnn-ai` | MNN AI | API key, aggregator | [link](https://mnnai.ru) | Free plan: $1 monthly credits, 10 RPM and access only to models marked Free. | +| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | +| `modelscope` | `ms` | ModelScope | API key | [link](https://modelscope.cn) | Free tier via ModelScope API-Inference — Alibaba account required. | +| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | ⚠️ **DEPRECATED.** Monster API shuttered operations on 2026-06-30. Use alternative OpenAI-compatible providers. | +| `moonshot` | `moonshot` | Kimi | API key | [link](https://platform.kimi.ai?aff=omniroute) | — | +| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | +| `muse-code` | `mc` | Muse Code (Meta) | API key | [link](https://github.com/meta-llama/llama-stack) | Use your META_API_KEY env var as a Bearer token. Muse Code CLI uses the OpenAI Responses API wire format (POST /responses). | +| `naga-ac` | `naga` | Naga.ac | API key, aggregator | [link](https://naga.ac) | Get API key at naga.ac — Google/GitHub/Discord signup available. | +| `naga-ai` | `naga-ai` | Naga AI | API key, aggregator | [link](https://naga.ac) | Models marked :free are publicly listed, but no numeric quota is confirmed. Naga's policy warns that free-tier prompts and outputs may be collected or used for training. | +| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | +| `nara` | `nara` | NaraRouter | API key | [link](https://bynara.id) | Get a free API key via NaraRouter's Telegram channel, then paste it here as a Bearer token. | +| `navy` | `navy` | NavyAI | API key | [link](https://api.navy) | Create a free API key from the NavyAI dashboard, then paste it here as a Bearer token. | +| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | +| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | +| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | +| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | +| `novita` | `novita` | Novita AI | API key, video, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | +| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | +| `nube` | `nube` | Nube.sh | API key | [link](https://nube.sh) | — | +| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | +| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | +| `ofoxai` | `ofoxai` | OfoxAI | API key, aggregator | [link](https://ofox.ai) | The current catalog advertises 10+ free models without a public numeric quota; review upstream provenance, retention and training terms before production use. | +| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/keys) | — | +| `openadapter` | `oad` | OpenAdapter | API key | [link](https://openadapter.dev) | Use your OpenAdapter API key in Authorization: Bearer sk-cv-. Fully OpenAI-compatible. API base URL: https://api.openadapter.in/v1. | +| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | +| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | +| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | +| `openference-api` | `ofa` | Openference API | API key | [link](https://openference.com) | Free plan: 3-day trial with open-source models — no credit card required | +| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | +| `openvecta` | `openvecta` | OpenVecta | API key | [link](https://openvecta.com) | Free credits on signup for OpenAI-compatible inference across LLMs, embeddings, and reasoning models | +| `orcarouter` | `orcarouter` | OrcaRouter | API key | [link](https://www.orcarouter.ai) | — | +| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | +| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | +| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | +| `pioneer` | `pn` | Pioneer AI | API key | [link](https://pioneer.ai) | $75 free usage credits — no credit card required | +| `plamo` | `plamo` | PLaMo | API key | [link](https://plamo.preferredai.jp/api) | — | +| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | +| `poixe-ai` | `poixe-ai` | Poixe AI | API key, aggregator | [link](https://poixe.com) | Current public free limits are small and model-group specific: 2 RPM/5 RPD for large-cup models and 20 RPM/50 RPD for small-cup models. | +| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | Anonymous/keyless access to the documented free models is best-effort. Local v3.8.50 verification (2026-07-31) returned 401 via OmniRoute and Cloudflare 1010 on direct upstream probes from the same network. Premium models still require a Pollinations API key from enter.pollinations.ai. | +| `poolside` | `poolside` | Poolside | API key | [link](https://poolside.ai) | Laguna S 2.1 and XS 2.1 are free during Preview; no public numeric quota is published. | +| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | ⚠️ **DEPRECATED.** serving.app.predibase.com no longer resolves (sweep 2026-06-19); the managed serving API appears discontinued. | +| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid | +| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product-s/qianfan_home) | — | +| `qiniu` | `qiniu` | Qiniu | API key | [link](https://www.qiniu.com) | — | +| `qwen-cloud` | `qwc` | Qwen Cloud | API key | [link](https://www.qwencloud.com/) | — | +| `qwen-cloud-token-plan` | `qct` | Qwen Cloud Token Plan | API key | [link](https://www.qwencloud.com/pricing/token-plan) | — | +| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | +| `regolo` | `regolo` | Regolo AI | API key | [link](https://regolo.ai) | Get your Regolo API key from regolo.ai, then paste it here as a Bearer token. | +| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | +| `requesty` | `requesty` | Requesty | API key | [link](https://requesty.ai) | Free tier ~200 requests/day - multi-model routing gateway (300+ models) | +| `routeway` | `routeway` | Routeway | API key | [link](https://routeway.ai) | Create a free API key at routeway.ai, then paste it here as a Bearer token. | +| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | +| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | +| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | +| `sarvam` | `sarvam` | Sarvam AI | API key | [link](https://docs.sarvam.ai) | ₹1,000 in free signup credits — never expire | +| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/docs/ai-data/generative-apis/) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | +| `sealion` | `sealion` | SEA-LION | API key | [link](https://sea-lion.ai) | Sign in at sea-lion.ai with Google (no card, no region wall), create an API key, then paste it here. | +| `segmind` | `segmind` | Segmind | API key, image, video | [link](https://segmind.com) | Use your Segmind API key in the x-api-key header. OmniRoute targets https://api.segmind.com/v1/ and returns the generated image/video bytes directly. | +| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | +| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus currently listed $0 models after identity verification; availability and limits may change | +| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | +| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `speka` | `speka` | Speka AI | API key, aggregator | [link](https://speka.me) | Free plan: $1 monthly usage, 10 RPM, one API key and access to open models and the playground; no card required. | +| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | +| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | +| `sumopod` | `sumopod` | SumoPod | API key | [link](https://ai.sumopod.com) | Use your SumoPod API key (sk-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://ai.sumopod.com/v1. | +| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | +| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | +| `tabitoken` | `tabitoken` | TabiToken | API key, aggregator | [link](https://tabitoken.com) | — | +| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | +| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | +| `tinyfish` | `tf` | TinyFish Fetch | API key | [link](https://docs.tinyfish.ai/fetch-api) | X-API-Key from agent.tinyfish.ai/api-keys | +| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | — | +| `token-kiosk` | `tk` | Token Kiosk | API key | [link](https://agent-router.gaib.ai) | Use your Token Kiosk API key in Authorization: Bearer . Fully OpenAI-compatible gateway. API base URL: https://agent-router.gaib.ai/v1. | +| `tokenreply` | `tokenreply` | TokenReply | API key, aggregator | [link](https://www.tokenreply.com) | Free-tagged models have model- and campaign-specific daily limits; no fixed global free quota is published. | +| `tokenrouter` | `trk` | TokenRouter | API key | [link](https://tokenrouter.com) | Use your TokenRouter API key in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.tokenrouter.com/v1. | +| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | +| `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. | +| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | +| `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. | +| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | +| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | +| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | +| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | +| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | +| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | +| `void-ai` | `void-ai` | Void AI | API key, aggregator | [link](https://voidai.app) | The public model catalog marks some models with a free plan requirement, but access is conditional and no numeric quota is confirmed. | +| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | +| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | +| `wafer` | `wafer` | Wafer AI | API key | [link](https://wafer.ai) | — | +| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | +| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | +| `writer` | `writer` | Writer | API key | [link](https://dev.writer.com) | — | +| `x5lab` | `x5lab` | X5Lab | API key | [link](https://x5lab.dev) | Use your X5Lab API key (x5-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.x5lab.dev/v1. | +| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | Use an official xAI API key, or sign in with xAI OAuth. Grok Build JWT sessions remain a separate provider. | +| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | +| `xiaomi-mimo-token-plan` | `mimotp` | Xiaomi MiMo Token Plan | API key | [link](https://mimo.mi.com) | — | +| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | +| `yolo-auto` | `yolo-auto` | Yolo-Auto | API key, aggregator | [link](https://yolo-auto.com) | Free API access is request-limited and intended for testing; no numeric daily quota is published and free access is not promised indefinitely. | +| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | +| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. | +| `zerolimitai` | `zerolimitai` | ZeroLimitAI | API key, aggregator | [link](https://www.zerolimitai.com) | Temporary free trial is advertised, but official pages conflict between 3 and 7 days; a 100-calls/day claim is not treated as permanent. | +| `zylo-api` | `zylo` | Zylo API | API key, aggregator | [link](https://zyloai.net) | Basic plan: 10 RPM, 7,200 requests/day and 200,000 tokens/day; limited to Basic text models. | + +## Local Providers (14) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | +| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | +| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | +| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | +| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | +| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | +| `mlx-gemma` | `mlx-gemma` | MLX Gemma 26B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11435. Requires uv and mlx-lm installed. Model: mlx-community/gemma-4-26B-A4B-it-qat-q4_0-mlx-aligned (~15.9GB peak memory). | +| `mlx-qwen` | `mlx-qwen` | MLX Qwen 3.8 27B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11436. Requires uv and mlx-lm installed. Model: maglun/Qwen3.8-27B-MLX-Mixed-3.80bpw (~13.1GB peak memory). | +| `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. | +| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | +| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | +| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | + +## Search Providers (13) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | +| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | +| `firecrawl` | `fc` | Firecrawl | Search | [link](https://firecrawl.dev) | API key from firecrawl.dev/app/api-keys (or set your self-hosted Firecrawl base URL) | +| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | +| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | +| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/keys) | Same API key as Ollama Cloud (from ollama.com/settings/keys) | +| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | +| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs/google) | API key from SearchAPI (query param or Bearer auth) | +| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | +| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | +| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | +| `x-search` | `x_search` | X Search (Grok) | Search | [link](https://docs.x.ai/developers/tools/x-search) | SuperGrok OAuth (xai-oauth) or xAI API key. This is Grok X Search, not the X Developer MCP. | +| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/business/api/) | X-API-Key from the You.com platform dashboard | + +## Audio-only Providers (12) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | +| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | +| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | +| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | +| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | +| `fishaudio` | `fishaudio` | Fish Audio | Audio | [link](https://fish.audio) | — | +| `gladia` | `gladia` | Gladia | Audio | [link](https://gladia.io) | — | +| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | +| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | +| `rev-ai` | `revai` | Rev AI | Audio | [link](https://www.rev.ai) | — | +| `soniox` | `sx` | Soniox | Audio | [link](https://soniox.com) | — | +| `speechmatics` | `sm` | Speechmatics | Audio | [link](https://www.speechmatics.com) | Free tier — 8 hours/month, no credit card required. Batch (async) mode only. | + +## Upstream Proxy Providers (2) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | +| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | + +## Cloud Agent Providers (3) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | +| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | +| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | + +## System Providers (1) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `auto` | `auto` | Auto (Zero-Config) | System | — | — | + +## Sources of truth + +- Catalog: [`src/shared/constants/providers.ts`](../../src/shared/constants/providers.ts) +- Registry (per-model details): [`open-sse/config/providerRegistry.ts`](../../open-sse/config/providerRegistry.ts) +- Executors: [`open-sse/executors/`](../../open-sse/executors/) (106 implementations) +- Translators: [`open-sse/translator/`](../../open-sse/translator/) + +## See Also + +- [FREE_TIERS.md](./FREE_TIERS.md) — curated free-tier guide +- [USER_GUIDE.md](../guides/USER_GUIDE.md) — provider setup walkthrough +- [ARCHITECTURE.md](../architecture/ARCHITECTURE.md) — overall architecture diff --git a/README.md b/README.md index fec2392ea7d0..6600b4064929 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 348 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 348 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 349 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 349 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. @@ -101,7 +101,7 @@ ⚙️ Features 🎯 Combos - 🌐 Providers + 🌐 Providers 🔌 CLI & MCP @@ -210,7 +210,7 @@ curl http://localhost:20128/v1/chat/completions \ -The Promise — One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 348 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 57 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests). +The Promise — One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 349 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests).

@@ -461,7 +461,7 @@ All **19** strategies — mix & match per combo step: -What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 348 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. +What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 349 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. 📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md) @@ -559,7 +559,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute - **🖼️ New endpoints** — `/v1/ocr` (Mistral OCR) and `/v1/audio/translations` (Whisper-style) round out the media surface. → [API Reference](docs/reference/API_REFERENCE.md) - **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md) - **🌍 Deployment & ops** — reverse-proxy `basePath`, browser-language auto-detect, per-key device tracking, root-less MITM trust, zh-TW localization. → [Environment](docs/reference/ENVIRONMENT.md) -- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **348-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) +- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **349-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) - **📡 Routing transparency** — every response carries an `X-OmniRoute-Decision` header naming the strategy/provider/latency that served it, a new `cache-optimized` combo strategy + Auto-Combo `cacheAffinity` factor route repeat requests back to the connection holding the cached prefix, and a read-only `/v1/auto-combo/{channel}/candidates` endpoint exposes an `auto/*` channel's live candidate pool. → [Auto-Combo](docs/routing/AUTO-COMBO.md) - **⚡ Local performance & infra** — one-click local Redis, Cloudflare Workers / Deno Deploy relay deployers, Bifrost & Mux as supervised embedded services. → [Embedded Services](docs/frameworks/EMBEDDED-SERVICES.md) @@ -642,11 +642,11 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
-## 🌐 348 AI Providers — 90+ Free +## 🌐 349 AI Providers — 90+ Free
-> The most complete catalog of any open-source router: **348 providers**, **90+ with a free tier**, **57 free forever**. +> The most complete catalog of any open-source router: **349 providers**, **90+ with a free tier**, **56 free forever**.
@@ -990,11 +990,11 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ `:latest` follows the highest **published** stable SemVer. It does not track git `main`. Pin `:X.Y.Z` for GitOps. See [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels).The image pins **`OMNIROUTE_MEMORY_MB=1024`**. That is enough for the dashboard and a light chat. **Coding agents** (`POST /v1/responses` from Claude Code, Codex, Grok, …) need a much larger V8 heap or the process `FATAL ERROR`s at ~12 GiB under two overlapping long contexts. Size the container above the heap (native buffers sit outside V8): -| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | -| --- | --- | --- | -| Dashboard / light chat | `1024` (image default) | ≥2 g | -| One coding agent | `8192` | ≥10 g | -| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | +| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | +| ----------------------------------- | ------------------------------- | ---------------------- | +| Dashboard / light chat | `1024` (image default) | ≥2 g | +| One coding agent | `8192` | ≥10 g | +| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | ```bash docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ @@ -1003,6 +1003,7 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ ``` Full table: [Docker Guide — runtime RAM](docs/guides/DOCKER_GUIDE.md#runtime-ram-for-coding-agents). + > **Pre-release Docker channel:** `diegosouzapw/omniroute:next` and > `diegosouzapw/omniroute:next-web` follow the current default `release/v*` > branch. These mutable tags are intended only for testing unreleased fixes and @@ -1200,7 +1201,7 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c RuntimeNode.js 22.x / 24.x LTS — >=22.22.2 <23 || >=24.0.0 <27 LanguageTypeScript 6.0 — 100% TypeScript across src/ and open-sse/ (zero any in core since v2.0) FrameworkNext.js 16 + React 19 + Tailwind CSS 4 - Databasebetter-sqlite3 (SQLite, WAL journaling) + LowDB (JSON legacy) — 120 domain modules, 157 migrations + Databasebetter-sqlite3 (SQLite, WAL journaling) + LowDB (JSON legacy) — 120 domain modules, 159 migrations MemorySQLite FTS5 full-text + int8-quantized vector embeddings, typed decay SchemasZod 4 — MCP tool I/O validation + API contracts ProtocolsMCP (stdio / HTTP / SSE) + A2A v0.3 (JSON-RPC 2.0 + SSE) diff --git a/bin/cli/commands/combo.mjs b/bin/cli/commands/combo.mjs index 554c1c1d838c..8d58cf73bd93 100644 --- a/bin/cli/commands/combo.mjs +++ b/bin/cli/commands/combo.mjs @@ -307,6 +307,12 @@ export async function runComboCreateCommand(name, strategy = "priority", opts = } const models = Array.isArray(opts.models) ? opts.models : []; + if (!models.length) { + console.error( + "combo create requires at least one target. Pass --models and/or repeat --model ." + ); + return 1; + } try { return await withRuntime(async ({ kind, api, db }) => { diff --git a/bin/cli/commands/oauth.mjs b/bin/cli/commands/oauth.mjs index 8bf547b2c00f..c9f8386d2b9d 100644 --- a/bin/cli/commands/oauth.mjs +++ b/bin/cli/commands/oauth.mjs @@ -228,20 +228,38 @@ async function runSocialFlow(def, opts) { async function runDeviceFlow(def, opts) { const providerKey = resolveBackendKey(def.id); - const startRes = await apiFetch(`/api/providers/${providerKey}/auth/start`, { - ...targetApiOptions(opts), - method: "POST", - }); + let startRes = await apiFetch(`/api/oauth/${providerKey}/device-code`, targetApiOptions(opts)); + if (!startRes.ok) { + startRes = await apiFetch(`/api/providers/${providerKey}/auth/start`, { + ...targetApiOptions(opts), + method: "POST", + }); + } if (!startRes.ok) { process.stderr.write(`Failed to start device flow: ${startRes.status}\n`); process.exit(1); } const start = await startRes.json(); - process.stdout.write( - `\nDevice code: ${start.userCode ?? start.user_code ?? ""}\nVisit: ${start.verificationUri ?? start.verification_uri}\n\n` - ); - if (opts.browser !== false) - await openBrowser(start.verificationUri ?? start.verification_uri ?? ""); + const userCode = start.userCode ?? start.user_code ?? ""; + const verificationUri = + start.verificationUriComplete ?? + start.verification_uri_complete ?? + start.verificationUri ?? + start.verification_uri ?? + start.authUrl ?? + start.url ?? + ""; + + if (userCode) { + process.stdout.write(`\nDevice code: ${userCode}\nVisit: ${verificationUri}\n\n`); + } else if (verificationUri) { + process.stdout.write(`\nVisit: ${verificationUri}\n\n`); + } else { + process.stdout.write(`\nAuthorization URL not available\n\n`); + } + + if (opts.browser !== false && verificationUri) + await openBrowser(verificationUri); process.stderr.write("Waiting for device authorization...\n"); const deadline = Date.now() + (opts.timeout ?? 300000); const intervalMs = (start.intervalMs ?? start.interval ?? 5) * 1000; diff --git a/changelog.d/features/10987-logfare-free-provider.md b/changelog.d/features/10987-logfare-free-provider.md new file mode 100644 index 000000000000..507a528411fd --- /dev/null +++ b/changelog.d/features/10987-logfare-free-provider.md @@ -0,0 +1 @@ +- **feat(providers):** add Logfare as a free OpenAI-compatible provider — dashboard card with a Free badge and request-logging disclosure (every prompt/completion is logged for research; opt out at logfare.ai/consent), live model discovery from `https://logfare.ai/v1/models` (20 models, 11 chat-capable: kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3…), full chat/streaming through the existing OpenAI-compatible path, the real Logfare logo on the card, and a listing in the free-tiers guide. ([#10987](https://github.com/diegosouzapw/OmniRoute/pull/10987)) diff --git a/changelog.d/features/11104-operator-error-rules.md b/changelog.d/features/11104-operator-error-rules.md new file mode 100644 index 000000000000..f31e78c01f3e --- /dev/null +++ b/changelog.d/features/11104-operator-error-rules.md @@ -0,0 +1 @@ +- **feat(providers):** let operators declare per-provider error rules through `settings.providerErrorRules` instead of patching the catalog — an operator-supplied rule for a provider is consulted before the built-in `providerRuleRegistry`, receives the raw error text, and has its declared scope/cooldown/reason actually honored end to end, for any provider (declaring the rule is the opt-in — no extra allowlist entry needed). Matches are plain case-insensitive substrings (never RegExp) and bounded to 50 rules to keep the hot path safe ([#11104](https://github.com/diegosouzapw/OmniRoute/pull/11104)) diff --git a/changelog.d/features/11190-usage-command-json.md b/changelog.d/features/11190-usage-command-json.md new file mode 100644 index 000000000000..d7655f04c51b --- /dev/null +++ b/changelog.d/features/11190-usage-command-json.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage` gains a structured form — `?format=json` returns the key's own usage as `ApiKeyUsageLimitStatus` + `UsageSnapshot` instead of `text/plain`. This is the surface a UI (the OmniCopilot panel) consumes to show a key holder their daily/weekly spend and quota reset. The route is self-service (the caller's own key, gated by `allowUsageCommand`), not the management surface; refusals come back as a discriminated `{ "allowed": false, "error": … }` so a UI can tell "not allowed" apart from "allowed but nothing cached yet". The endpoint was previously undocumented in `API_REFERENCE.md`; it now has a section ([#11190](https://github.com/diegosouzapw/OmniRoute/pull/11190)) diff --git a/changelog.d/features/11192-usage-command-providers-array.md b/changelog.d/features/11192-usage-command-providers-array.md new file mode 100644 index 000000000000..b7ef42110913 --- /dev/null +++ b/changelog.d/features/11192-usage-command-providers-array.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage?format=json` now returns `providers[]` — every connection's quota snapshot, not just the single selected one — so a panel can render Codex / Claude / OpenCode side by side. The collector already gathered all of them; the single-pick `provider` field (kept) is a terminal presentation choice. Closes the per-connection gap from OmniCopilot #8 ([#11192](https://github.com/diegosouzapw/OmniRoute/pull/11192)) diff --git a/changelog.d/fixes/10550-responses-reasoning-transport.md b/changelog.d/fixes/10550-responses-reasoning-transport.md index d34c433debd8..e2b40cdb8c97 100644 --- a/changelog.d/fixes/10550-responses-reasoning-transport.md +++ b/changelog.d/fixes/10550-responses-reasoning-transport.md @@ -1 +1 @@ -- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Combos now drop incompatible continuation reasoning by default and can explicitly skip incompatible targets, while known providers no longer show redundant encrypted-reasoning controls. (#10550) +- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Direct requests drop incompatible continuation reasoning by default; combos can explicitly skip incompatible targets without mutating the request. Known providers no longer show redundant encrypted-reasoning controls. (#10550, #10959) diff --git a/changelog.d/fixes/10949-mixed-reasoning-plaintext.md b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md new file mode 100644 index 000000000000..05a055ec5384 --- /dev/null +++ b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md @@ -0,0 +1 @@ +- Preserve explicit plaintext reasoning when a Responses reasoning item also carries opaque provider state (rare OpenCode Go `deepseek-v4-flash` responses). Mixed plaintext + opaque input is projected onto the target transport: plaintext targets keep portable text, opaque targets keep provider state. Opaque-only reasoning is dropped when the selected target cannot replay it, allowing cross-model conversations to continue. (#10949, #10959) diff --git a/changelog.d/fixes/11015-shutdown-track-sse.md b/changelog.d/fixes/11015-shutdown-track-sse.md new file mode 100644 index 000000000000..1ed99b3669d8 --- /dev/null +++ b/changelog.d/fixes/11015-shutdown-track-sse.md @@ -0,0 +1 @@ +- **fix(resilience):** count heavyweight `/v1` admission leases in the SIGTERM drain and send `Retry-After` on shutdown 503s so Recreate no longer looks like an empty 502 ([#11015](https://github.com/diegosouzapw/OmniRoute/issues/11015)) — thanks @RaviTharuma diff --git a/changelog.d/fixes/11060-perplexity-filter.md b/changelog.d/fixes/11060-perplexity-filter.md new file mode 100644 index 000000000000..c221d3ccaba1 --- /dev/null +++ b/changelog.d/fixes/11060-perplexity-filter.md @@ -0,0 +1 @@ +- fix(providers): filter Perplexity model import to the Sonar family so Agent-API catalog ids stop surfacing as routable chat models (#11060) diff --git a/changelog.d/fixes/11085-claude-code-tool-name-casing.md b/changelog.d/fixes/11085-claude-code-tool-name-casing.md new file mode 100644 index 000000000000..5ad424c14182 --- /dev/null +++ b/changelog.d/fixes/11085-claude-code-tool-name-casing.md @@ -0,0 +1 @@ +- **fix(claude):** restore canonical tool names (`bash` → `Bash`, `croncreate` → `CronCreate`) on non-streaming OpenAI→Claude conversion and through identity-echo alias maps, so Claude Code stops rejecting tool calls with "No such tool available" ([#11085](https://github.com/diegosouzapw/OmniRoute/pull/11085)) — thanks @linhdmn diff --git a/changelog.d/fixes/11095-termux-onnx.md b/changelog.d/fixes/11095-termux-onnx.md new file mode 100644 index 000000000000..8c268037104c --- /dev/null +++ b/changelog.d/fixes/11095-termux-onnx.md @@ -0,0 +1 @@ +- fix(install): make the ONNX dependency chain optional so Termux/Android installs succeed again (#11095) diff --git a/changelog.d/fixes/11101-reject-silent-validation.md b/changelog.d/fixes/11101-reject-silent-validation.md new file mode 100644 index 000000000000..04b2a67d5a07 --- /dev/null +++ b/changelog.d/fixes/11101-reject-silent-validation.md @@ -0,0 +1 @@ +- **fix(providers):** Reject silent validation degradation on provider connection patch — unknown `rateLimitOverrides` keys (e.g. a typo'd `tpm`) and empty/non-numeric values now return `400` with the rejected key list instead of being silently dropped ([#11101](https://github.com/diegosouzapw/OmniRoute/pull/11101)) diff --git a/changelog.d/fixes/11102-combo-suggestion-count.md b/changelog.d/fixes/11102-combo-suggestion-count.md new file mode 100644 index 000000000000..3cbf6f11d3de --- /dev/null +++ b/changelog.d/fixes/11102-combo-suggestion-count.md @@ -0,0 +1 @@ +- **Autopilot suggestion counter:** the combo health autopilot summary now reports `suggestionCount` (the real number of suggested actions across all issues) instead of conflating it with link counts, while keeping `actionableCount` as a deprecated alias for backward compatibility. The `run_combo_test` action now links to the dashboard with the combo id (`/dashboard/combos?test=`) rather than the read-only API route, so operators can actually trigger a test from the UI ([#11102](https://github.com/diegosouzapw/OmniRoute/pull/11102)). diff --git a/changelog.d/fixes/11103-persist-config-audit-log.md b/changelog.d/fixes/11103-persist-config-audit-log.md new file mode 100644 index 000000000000..aeb53b178145 --- /dev/null +++ b/changelog.d/fixes/11103-persist-config-audit-log.md @@ -0,0 +1 @@ +- **Config audit persistence:** persist the configuration audit trail to SQLite (`config_audit_log`) instead of an in-memory buffer capped at 1000 volatile entries, and bound its growth with `cleanupConfigAudit()` driven by the `retention.configAudit` setting (default 30 days), wired into `runAutoCleanup` ([#11103](https://github.com/diegosouzapw/OmniRoute/pull/11103)). diff --git a/changelog.d/fixes/11109-stream-recovery-toolcall.md b/changelog.d/fixes/11109-stream-recovery-toolcall.md new file mode 100644 index 000000000000..04a43382e427 --- /dev/null +++ b/changelog.d/fixes/11109-stream-recovery-toolcall.md @@ -0,0 +1 @@ +- fix(sse): resume mid-stream recovery after a _completed_ tool call — `finish_reason: "tool_calls"` is now tracked per-call instead of as a general terminal marker, so truncation of trailing prose after a fully-delivered tool call is recoverable while in-flight calls stay blocked ([#11109](https://github.com/diegosouzapw/OmniRoute/pull/11109)) diff --git a/changelog.d/fixes/11116-reasoning-effort-capability-discovery.md b/changelog.d/fixes/11116-reasoning-effort-capability-discovery.md new file mode 100644 index 000000000000..fbc4dfe694e3 --- /dev/null +++ b/changelog.d/fixes/11116-reasoning-effort-capability-discovery.md @@ -0,0 +1 @@ +- **fix(providers):** `reasoning_effort` now learns the accepted values from a provider's own 400/422 response and clamps to the highest one instead of forwarding an unsupported `xhigh`/`max` (or a hardcoded `"high"` fallback) — fixes custom OpenAI-compatible connections and registered providers with no reasoning metadata ([#11116](https://github.com/diegosouzapw/OmniRoute/pull/11116)) — thanks @maxmad64bis diff --git a/changelog.d/fixes/11144-responses-parallel-tool-calls-index.md b/changelog.d/fixes/11144-responses-parallel-tool-calls-index.md new file mode 100644 index 000000000000..35ba19b27960 --- /dev/null +++ b/changelog.d/fixes/11144-responses-parallel-tool-calls-index.md @@ -0,0 +1 @@ +- **fix(sse):** parallel `function_call` items in a Responses API stream (e.g. several tool calls dispatched in the same turn) now each get a stable, distinct `index`/`id` when translated to Chat Completions streaming deltas, instead of colliding on index 0 and tripping strict stream parsers with `Expected 'id' to be a string.` ([#11144](https://github.com/diegosouzapw/OmniRoute/pull/11144)) diff --git a/changelog.d/fixes/11154-provider-registry-node-net-bundle.md b/changelog.d/fixes/11154-provider-registry-node-net-bundle.md new file mode 100644 index 000000000000..b30d9e139265 --- /dev/null +++ b/changelog.d/fixes/11154-provider-registry-node-net-bundle.md @@ -0,0 +1 @@ +- fix(dashboard): keep `open-sse/config/providerRegistry.ts` free of `node:net` so the provider detail client bundle builds again — the host classification moved to a platform-free `src/shared/network/privateHost.ts` with a pure-JS `isIP` equivalent, leaving the #11122 routing behaviour unchanged (#11154) diff --git a/changelog.d/fixes/11162-combo-create-requires-model.md b/changelog.d/fixes/11162-combo-create-requires-model.md new file mode 100644 index 000000000000..228e6a9b20f8 --- /dev/null +++ b/changelog.d/fixes/11162-combo-create-requires-model.md @@ -0,0 +1 @@ +- **Combo create:** creating a routing combo without any model is now refused (`400`) — the CLI requires `--models`/`--model` on `combo create`, matching the dashboard which already rejected empty combos. diff --git a/changelog.d/fixes/11165-shared-registry-passthrough-model-lockout.md b/changelog.d/fixes/11165-shared-registry-passthrough-model-lockout.md new file mode 100644 index 000000000000..eabe67cb09cb --- /dev/null +++ b/changelog.d/fixes/11165-shared-registry-passthrough-model-lockout.md @@ -0,0 +1 @@ +- **fix(resilience):** a missing-model `404` on a provider that declares `passthroughModels: true` in the shared registry (novita, uncloseai, orcarouter and 37 others) now locks out only that model instead of cooling the entire connection — `hasPerModelQuota()` previously read only the open-sse registry and the local/self-hosted families ([#11165](https://github.com/diegosouzapw/OmniRoute/pull/11165)) — thanks @yourspraveen diff --git a/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md new file mode 100644 index 000000000000..82a88c5905a7 --- /dev/null +++ b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md @@ -0,0 +1 @@ +- **fix(executors):** OpencodeExecutor rotates (or retries once on a single-account direct path) on upstream 400 empty-body rejections — malformed completion envelopes with no error field were propagated as success and killed client sessions. Bounded +1 attempt per request; body reads are conditioned on status 400 so successful/streaming responses are never buffered. 400s carrying an error field keep propagating immediately. diff --git a/changelog.d/fixes/release-v3850-basereds-tests-i18n.md b/changelog.d/fixes/release-v3850-basereds-tests-i18n.md new file mode 100644 index 000000000000..3a6dee61b9cd --- /dev/null +++ b/changelog.d/fixes/release-v3850-basereds-tests-i18n.md @@ -0,0 +1 @@ +- fix(i18n): complete Vietnamese translations for recently added UI strings (#9985) diff --git a/changelog.d/maintenance/11053-stryker-oauth-autoimport-registration.md b/changelog.d/maintenance/11053-stryker-oauth-autoimport-registration.md new file mode 100644 index 000000000000..b2e214190099 --- /dev/null +++ b/changelog.d/maintenance/11053-stryker-oauth-autoimport-registration.md @@ -0,0 +1 @@ +- fix(quality): register `tests/unit/authz/oauth-autoimport-local-only.test.ts` in stryker `tap.testFiles` (residual of #11053) diff --git a/changelog.d/maintenance/11160-drain-v3850-basereds-docs-counts-orphan-test.md b/changelog.d/maintenance/11160-drain-v3850-basereds-docs-counts-orphan-test.md new file mode 100644 index 000000000000..e695a2b8fc0b --- /dev/null +++ b/changelog.d/maintenance/11160-drain-v3850-basereds-docs-counts-orphan-test.md @@ -0,0 +1 @@ +- chore(quality): drain two `release/v3.8.50` base-reds — refresh the drifted doc counts (159 migrations, 56 free-forever providers, 40 free-tier pools, incl. the 42 `llm.txt` locale mirrors) and move `uncloseai-noauth.test.ts` to a collected path so the UncloseAI no-auth regression guard actually runs (#11160) diff --git a/changelog.d/maintenance/vi-harimport-parity.md b/changelog.d/maintenance/vi-harimport-parity.md new file mode 100644 index 000000000000..b08b8dc92f7d --- /dev/null +++ b/changelog.d/maintenance/vi-harimport-parity.md @@ -0,0 +1 @@ +- fix(i18n): translate the 14 `providers.harImport*` keys into Vietnamese (parity gap left by #11069) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 536318240a40..2eb468e16c6d 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,5 +1,6 @@ { "_rebaseline_2026_08_20_10531_freebuff_provider": "PR #10531 (adrianaryaputra, feat/freebuff-provider-support, closes #6793) own growth: src/shared/constants/providers/apikey/gateways.ts 1283->1298 (+15, the freebuff APIKEY_PROVIDERS_GATEWAYS catalog entry, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines) and src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx 1062->1067 (+5, freebuff credential placeholder/hint at the existing per-provider switch chokepoint). Covered by tests/unit/freebuff-provider.test.ts (9/9 passing).", + "_rebaseline_2026_08_21_10987_logfare_provider": "PR #10987 (jonlwheat2-gif, feat/10644-logfare-provider, closes #10644) own growth: src/shared/constants/providers/apikey/gateways.ts 1298->1321 (+23, the logfare APIKEY_PROVIDERS_GATEWAYS catalog entry with Free badge/freeNote/apiHint documenting the request-logging policy, additive data at the existing registry chokepoint, same god-file no-split rationale as the prior gateways.ts rebaselines: #10531 freebuff, merge-storm 2026-08-11). Covered by tests/unit/logfare-registry.test.ts (1/1 passing).", "_rebaseline_2026_08_20_10574_reasoning_transport_fallback": "PR #10574 (jackjinke, fix/responses-reasoning-transport, fixes #10550) own growth: src/sse/handlers/chatHelpers.ts 1017->1019 (+2 = the new reasoningTransportFallback option threaded through executeChatWithBreaker's options destructure and its downstream handleSingleModel call, at the existing per-attempt options-passthrough chokepoint; not extractable without splitting the option-forwarding call itself). Covered by the PR's own reasoning-policy test suite (tests/unit/chatcore-translation-paths.test.ts, tests/unit/combo-attempt-body-isolation-7847.test.ts, tests/unit/reasoning-cache.test.ts, tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts among others), 446/446 focused tests passing.", "_rebaseline_2026_08_18_10517_zed_hosted_oauth_callback_port": "PR #10517 (phatchau036, fix/zed-hosted-oauth-callback-port) own growth: src/shared/components/OAuthModal.tsx 1131->1148 (wc -l; check-file-size.mjs counts via split(\"\\n\").length so the gate sees 1134->1149, +15/+18, crosses the frozen 1134 cap). Wires the zed-hosted native-app callback auto-complete: forceManual gating on isTrueLocalhost for zed-hosted, the loopback-redirect-URI comment block, and the exchangeToken full-URL-as-code branch, all at the existing provider-switch chokepoints this modal already carries growth for (seventh bump: 969->989->993->998->1030->1056->1100->1149; structural shrink tracked in #3501). The actual port-derivation logic lives in src/lib/oauth/providers/zed-hosted.ts (not frozen here) and was hardened during pre-merge review to use the server's own getRuntimePorts() instead of a browser-guessed scheme/port, covered by the new tests/unit/zed-hosted-loopback-port-derivation.test.ts (8/8 passing).", "_rebaseline_2026_08_13_10243_codex_fingerprint_merge": "PR #10243 (xz-dev, Codex OAuth fingerprint convergence) merge into release/v3.8.50: src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts crossed the 1000-line new-file cap for the first time (974 on base, 997 on the PR's own branch, 1013 after merging + prettier reflow) purely from combining two independent, already-legitimate feature additions that landed on the same shared UI-helper file — this PR's own Codex fingerprint-mode select/toggle wiring (CODEX_FINGERPRINT_MODE_VALUES, getCodexFingerprintModeLabel, CodexFingerprintModeValue) plus #8949's unrelated Codex account-service-tier helpers merged concurrently on release/v3.8.50. Neither addition alone crosses the cap; git's line-level auto-merge does not detect a threshold crossing. Not modularized as part of this conflict-resolution merge commit (out of scope — this is a merge, not a feature change). Covered by the PR's own tests/unit/codex-fingerprint-convergence.test.ts, tests/unit/executor-codex.test.ts, tests/unit/provider-specific-data-schema.test.ts (all passing post-merge).", @@ -388,6 +389,10 @@ "open-sse/services/claudeCodeCompatible.ts": 1563, "open-sse/services/combo.ts": 4742, "open-sse/services/compression/strategySelector.ts": 1379, + "open-sse/services/compression/engines/ccr/index.ts": 1024, + "_rebaseline_2026_08_22_11084_ccr_caller_gate": "PR #11084 (HouMinXi) own growth: open-sse/services/compression/engines/ccr/index.ts 1000->1024 (first listing — the engine was unlisted and drifted just over the 1000 cap; +24 are the callerSupportsCcrRetrieve gate that skips replacement entirely for callers without the retrieve tool, closing the stranded-prompt incident measured in production). Covered by tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", + "open-sse/services/contextManager.ts": 1001, + "_rebaseline_2026_08_22_11113_purify_system_first": "PR #11113 (ggdayup) own growth: open-sse/services/contextManager.ts 1000->1001 (+1, purifyHistory merges the compression notice into the leading system message instead of splicing a second one mid-array — live-confirmed TokenRouter 400s; the +1 is the merge-into-leading branch, not extractable). Covered by tests/unit/context-manager-purify-system-first.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "open-sse/services/rateLimitManager.ts": 1517, "open-sse/translator/response/openai-responses.ts": 1652, "open-sse/utils/cursorAgentProtobuf.ts": 1956, @@ -443,10 +448,11 @@ "src/shared/components/ModelSelectModal.tsx": 1138, "src/shared/constants/providers/apikey/gateways.ts": 1250 }, - "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1080, + "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1082, + "_rebaseline_2026_08_22_11156_enter_check_disabled": "PR #11156 (rqzbeh) own growth: AddApiKeyModal.tsx 1080->1082 (+2, Enter keydown handler now mirrors the isCheckDisabled condition — owner-requested post-merge polish from #11056; the rest of the diff is Prettier reflow). Covered by tests/unit/ui/add-api-key-modal-enter-key.test.tsx (jsdom render test, Enter dispatch assertions).", "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051, "src/shared/components/ModelSelectModal.tsx": 1138, - "src/shared/constants/providers/apikey/gateways.ts": 1298, + "src/shared/constants/providers/apikey/gateways.ts": 1321, "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1387, "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry": "DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legitima acima do cap; gateways.ts = god-file de catalogo de providers que cresceu com PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o proprio PR #9421 quebrou o arquivo); bridge.ts (PR #8949) = ponte Chromium vendored; proxyFetch.ts 1207->1220 = drift herdado de merges. Owner autorizou rebaseline com anotacao (2026-08-11).", "src/lib/modelCapabilities.ts": 1072, @@ -454,7 +460,8 @@ "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 1014, "open-sse/config/imageRegistry.ts": 1034, "src/sse/handlers/chatHelpers.ts": 1019, - "src/shared/middleware/chatBodyAdmission.ts": 1005, + "src/shared/middleware/chatBodyAdmission.ts": 1009, + "_rebaseline_2026_08_22_11020_sigterm_drain": "PR #11020 (RaviTharuma) own growth: chatBodyAdmission.ts 1005->1009 (+4, heavyweight admission leases now increment the SIGTERM drain counter and releaseChatAdmissionWhenDone holds it for the SSE lifetime — closes #11015; +4 are the lease/drain wiring lines at the existing admission chokepoint). Covered by tests/unit/chat-body-admission.test.ts heavyweight-lease cases. Owner pre-authorized baseline bumps 2026-08-22.", "_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file).", "open-sse/executors/commandCode.ts": 1059, "_rebaseline_2026_08_21_10859_vision_bridge_catalog": "#10859 own growth (Vision Bridge fixes #10808/#10809): src/lib/modelCapabilities.ts 1006->1016 (+10, cmd/gpt-5.3-codex* text-only capability resolution) and open-sse/executors/commandCode.ts 988->1023 (+35, Command Code wire-model normalization for bare ids + reasoning field fallback for opencode-routed gateways). Cohesive bug fixes at the existing capability-resolution / executor chokepoints; not extractable mid-fix. Covered by tests/unit/model-capabilities-command-code-codex-textonly-10703.test.ts, tests/unit/command-code-vision.test.ts, tests/unit/opencode-mimo-reasoning-details-nonstream.test.ts. Pushed directly to release (own-session miss: the original rebaseline was made in a throwaway validation worktree and never landed on the PR branch or the release before merge).", diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index 95e35a5adec2..604a5a8e19da 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -197,10 +197,11 @@ "_rebaseline_2026_08_09_v3850_release_close": "7666 -> 8045 (+379 gzip bytes, +4.9%). Release v3.8.50 close reconciliation measured twice with the real size-limit + @size-limit/file path on tip e0ce95c592. Per-entry measurements remain below their absolute budgets: omniroute.mjs 4380/15000, mcp-server.mjs 1195/5000, nodeRuntimeSupport.mjs 887/8000, reset-password.mjs 1583/6000. The growth accumulated through legitimate CLI/runtime work in this cycle, including global-install ESM alias resolution, Termux cache preparation, and MCP stdio startup hardening; no entrypoint is near its absolute ceiling. The direction:down ratchet stays blocking from this exact measured tip." }, "openapiBreaking": { - "value": 0, + "value": 4, "direction": "down", "dedicatedGate": true, - "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec." + "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec.", + "_rebaseline_2026_08_22_combo_create_min1": "0 -> 4, split 3 own + 1 inherited. Docs-only alignment of components.schemas.ComboCreate with the request contract already enforced by the API since 638fc5fbd (combo create refuses an empty model list) and d5034ea52: `model`/`nodes` were phantom properties the server never accepted, and `models` (array, minItems 1) is the real required field. OWN findings (3, caused by this commit): removed `model`, removed `nodes`, added required `models` on POST /api/combos — spec-vs-server drift, not client-facing breakage, no working client could have relied on the removed shapes. INHERITED finding (1, NOT caused by this PR's code changes — pre-existing drift already present at parent d5034ea52): PATCH /api/combos/{id} request-body-added-required; that route's patch operation declares its own inline requestBody (required: true, bare object schema, docs/openapi.yaml ~2107-2118) and does not reference ComboCreate, so this finding exists independently of the ComboCreate alignment (same own-growth vs inherited-drift convention as _rebaseline_2026_07_20_aliasresolver_hook_split_7808). No code change in this PR; follow-up tracking = this change's PR description." }, "mutationScore.src/sse/services/auth.ts": { "value": 52.57, diff --git a/docs/architecture/RESILIENCE_GUIDE.md b/docs/architecture/RESILIENCE_GUIDE.md index 030f65dd4141..0048761e6093 100644 --- a/docs/architecture/RESILIENCE_GUIDE.md +++ b/docs/architecture/RESILIENCE_GUIDE.md @@ -448,14 +448,14 @@ classification rules pick the fallback `reason` and lock `scope` Classification rules only see full error **text** (needed to match body markers like `额度不足`) for providers listed in the `FULL_TEXT_RULE_PROVIDERS` allowlist in `providerErrorRules.ts` — currently only `"agentrouter"`. For -every other provider, `checkFallbackError` hands `getProviderErrorRuleMatch` -only the structured error (`{code, type}`), which is enough for -header/status/code-based rules but blind to body-text markers. The helper -`resolveRuleMatchBody()` performs this selection: full error text for -allowlisted providers, the structured error otherwise. Adding a provider to -`FULL_TEXT_RULE_PROVIDERS` is an explicit per-provider opt-in — it exists so -that the default path for every provider not on the list stays -byte-for-byte unchanged. +every other **built-in catalog** provider, `checkFallbackError` hands +`getProviderErrorRuleMatch` only the structured error (`{code, type}`), which +is enough for header/status/code-based rules but blind to body-text markers. +The helper `resolveRuleMatchBody()` performs this selection: full error text +for allowlisted providers, the structured error otherwise. Adding a +**built-in** provider to `FULL_TEXT_RULE_PROVIDERS` is an explicit per-provider +opt-in — it exists so that the default path for every provider not on the +list stays byte-for-byte unchanged. A rule's `scope` (`model` / `provider` / `connection`) is a separate opt-in from `FULL_TEXT_RULE_PROVIDERS`: `checkFallbackError` only surfaces it as @@ -466,6 +466,31 @@ honorsRuleLockScope()` — today only `"agentrouter"`). See "Restated quota errors" above for what a `scope: "connection"` match actually does once a provider is on that allowlist. +**#11104 — operator-declared rules bypass both allowlists.** An operator can +declare a per-provider rule at runtime via `settings.providerErrorRules` +(`open-sse/config/providerErrorRules.ts::setOperatorProviderErrorRules`) +without editing this file. Gating an operator rule behind +`FULL_TEXT_RULE_PROVIDERS`/`HONORS_RULE_LOCK_SCOPE_PROVIDERS` — allowlists +meant to protect the **default** behavior of built-in catalog rules — would +make the settings mechanism inert for every provider except the ones already +listed there, since declaring the rule is already the operator's explicit +opt-in. `resolveRuleMatchBody()` and `honorsRuleLockScope()` both check +`hasOperatorRuleForProvider()` first: a provider with an operator rule gets +the raw error text and has its declared `scope` honored, regardless of +whether it also appears in either allowlist. + +**Known gap — `providerRuleRegistry` is never consulted for HTTP 400.** +`checkFallbackError`'s `BAD_REQUEST` branch classifies status 400 entirely +through its own pattern arrays (`MODEL_ACCESS_DENIED_PATTERNS`, +`CONTEXT_OVERFLOW_PATTERNS`, etc. in `accountFallback.ts`) and returns before +the `configuredRule`/`getProviderErrorRuleMatch` branch above it is reached. +A built-in catalog rule (or an operator rule) with `status: 400` is +syntactically valid but will never fire. No existing rule targets 400 today, +so nothing in production is affected — but a future 400 rule needs this +branch touched first, which is a larger change than adding a rule (it +reclassifies 400 for every provider already relying on the pattern-array +behavior) and is out of scope for a single-provider rule addition. + ### Adding a new quota-misstating gateway 1. Register one rule array in `statusRestatementRegistry` diff --git a/docs/changelog/fragments/10962.md b/docs/changelog/fragments/10962.md new file mode 100644 index 000000000000..5170a415e3f1 --- /dev/null +++ b/docs/changelog/fragments/10962.md @@ -0,0 +1 @@ +fix(catalog): expose only provider-routable GLM reasoning-effort tiers and remove unroutable ZCode aliases diff --git a/docs/diagrams/cli-terminal.svg b/docs/diagrams/cli-terminal.svg index e2ad57c8b13f..6ec68793d0c0 100644 --- a/docs/diagrams/cli-terminal.svg +++ b/docs/diagrams/cli-terminal.svg @@ -1,4 +1,4 @@ - + Compact animated terminal cycling three real OmniRoute CLI commands with a typewriter effect and a scrolling subcommand ticker; the first frame shows the completed providers-list screen. diff --git a/docs/diagrams/comparison-table.svg b/docs/diagrams/comparison-table.svg index 053194678cc8..f61e7ca0adde 100644 --- a/docs/diagrams/comparison-table.svg +++ b/docs/diagrams/comparison-table.svg @@ -1,4 +1,4 @@ - + Static-header comparison table where each capability row fades in top to bottom; the OmniRoute column is highlighted and shows a check or a leading value in every row, while competitors show a mix of checks, partials and crosses. diff --git a/docs/diagrams/free-tier-budget.svg b/docs/diagrams/free-tier-budget.svg index e4503b5534e7..393ee005941d 100644 --- a/docs/diagrams/free-tier-budget.svg +++ b/docs/diagrams/free-tier-budget.svg @@ -1,4 +1,4 @@ - + @@ -63,7 +63,7 @@ ~1.51B FREE TOKENS / MONTH · STEADY up to ~2.13B in your first month — signup credits - documented free tiers · 41 provider pools · 495 models · one endpoint + documented free tiers · 40 provider pools · 495 models · one endpoint diff --git a/docs/diagrams/promise-pillars.svg b/docs/diagrams/promise-pillars.svg index 761df36f2e88..aebefefabbfd 100644 --- a/docs/diagrams/promise-pillars.svg +++ b/docs/diagrams/promise-pillars.svg @@ -1,4 +1,4 @@ - + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. @@ -21,7 +21,7 @@ - One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. + One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. @@ -38,7 +38,7 @@ Never hit limits - Auto-fallback across 348 providers in + Auto-fallback across 349 providers in milliseconds. Quota out? The next provider takes over — zero downtime. diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index 99543d2471ea..3448cddc7ff7 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -28,7 +28,7 @@ Never stop coding. - Every AI tool → 348 providers — 90+ free — through one endpoint. + Every AI tool → 349 providers — 90+ free — through one endpoint. Claude Code · Codex · Cursor · Cline · Copilot · Antigravity  →  FREE Claude / GPT / Gemini · auto-fallback diff --git a/docs/getting-started/FREE-TIERS-GUIDE.md b/docs/getting-started/FREE-TIERS-GUIDE.md index 008fe9625778..6fd9dcc35bdb 100644 --- a/docs/getting-started/FREE-TIERS-GUIDE.md +++ b/docs/getting-started/FREE-TIERS-GUIDE.md @@ -26,6 +26,7 @@ These providers have a recurring, keyless, or uncapped free-access path in the a | **Kiro AI** | Claude Sonnet 4.5, Haiku 4.5, DeepSeek V3.2, and others | Audited catalog estimates a 25K-token shared monthly pool | OAuth/account flow; ToS flagged `avoid` in the catalog | | **OpenCode Free** | Current `*-free` model set in the provider registry | Keyless; no published token cap | No provider credential; ToS flagged `avoid` | | **Pollinations** | Current keyless model set; some former models are discontinued or key-required | Keyless; no published token cap | No provider credential for the keyless models | +| **Logfare** | kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3, and more | Free API key (no rate limits, no card); **every request is logged** for research (opt out at logfare.ai/consent) | Instant key at logfare.ai/register; ToS/privacy at logfare.ai/tos and logfare.ai/privacy | | **Cloudflare AI** | Workers AI catalog | Audited pool estimates ~30M tokens/month from published usage units | Cloudflare account and API credentials | | **Gemini** | Gemini Flash family | Audited pool estimates ~60M tokens/month | Google AI Studio API key; rate limits apply | | **Groq** | Llama, GPT-OSS, and Qwen models | Audited pool estimates ~15M tokens/month | Groq API key; rate limits apply | diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 61d414de2720..b24e7b7f6109 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -505,7 +505,7 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. | Constraint | Consequence | | --- | --- | | Single writer | Do **not** run multiple replicas against the same SQLite file. That corrupts the DB. | -| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. | +| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. New requests during the empty-endpoint window get a reverse-proxy **`502 Bad Gateway: Unknown error`**, not OmniRoute JSON — clients cannot distinguish this from a provider failure (#11015). | | Same event loop as `/healthz` | A busy catalog or compression tick can delay probes; a short timeout then restarts the **only** replica. | **Probe matrix** (see also [Kubernetes probe recommendations](../ops/MONITORING_GUIDE.md#kubernetes-probe-recommendations)): @@ -518,6 +518,35 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. **Upgrades:** expect every session to drop. Drain clients if you can; there is no rolling update on default SQLite. Compose `restart: unless-stopped` plus Docker `HEALTHCHECK` will also replace the only process when the container is Unhealthy — same blast radius. +Kubernetes snippet for a **single replica** (Recreate is required; do not raise `replicas` against one SQLite file): + +```yaml +spec: + replicas: 1 + strategy: + type: Recreate + template: + spec: + terminationGracePeriodSeconds: 90 + containers: + - name: omniroute + lifecycle: + preStop: + exec: + command: ["/bin/sleep", "15"] + readinessProbe: + httpGet: + path: /healthz + port: 20128 + periodSeconds: 5 + livenessProbe: + tcpSocket: + port: 20128 + periodSeconds: 20 +``` + +`preStop` sleep lets kube drop Service endpoints before SIGTERM so **new** traffic stops hitting the dying process. In-flight `/v1/responses` SSE is drained up to `SHUTDOWN_TIMEOUT_MS` (default 30s) via heavyweight admission leases (#11015). New requests that still reach the process get `503` + `Retry-After: 5`. The Recreate empty-endpoint gap until the replacement is Ready remains a hard outage — that is the SQLite topology, not a probe misconfig. + External Postgres / multi-writer HA is **not** a documented stock path. If you need HA, keep a single replica or run a topology the project has tested and documented separately. The Postgres/MySQL work lives in [#8075](https://github.com/diegosouzapw/OmniRoute/issues/8075). Until that ships, the only supported way to multiply **large** `/v1/responses` capacity is N independent processes (next section), not `replicas > 1` on one volume. ## Scale-out: N independent processes diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt index 4bbfed2b1a93..23e919c62807 100644 --- a/docs/i18n/ar/llm.txt +++ b/docs/i18n/ar/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/az/llm.txt b/docs/i18n/az/llm.txt index 22c327b815ff..fcb2eb38907a 100644 --- a/docs/i18n/az/llm.txt +++ b/docs/i18n/az/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt index 22c327b815ff..fcb2eb38907a 100644 --- a/docs/i18n/bg/llm.txt +++ b/docs/i18n/bg/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/bn/llm.txt b/docs/i18n/bn/llm.txt index 5435e521e9fa..f15efad72be9 100644 --- a/docs/i18n/bn/llm.txt +++ b/docs/i18n/bn/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt index db22c036bc1e..5c4acb63f9ae 100644 --- a/docs/i18n/cs/llm.txt +++ b/docs/i18n/cs/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt index b0b593980fd6..e4ca44f80cad 100644 --- a/docs/i18n/da/llm.txt +++ b/docs/i18n/da/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt index 118c84fed59a..833ca47bf6b7 100644 --- a/docs/i18n/de/llm.txt +++ b/docs/i18n/de/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt index c9e3f0f3b4d6..332a3a478904 100644 --- a/docs/i18n/es/llm.txt +++ b/docs/i18n/es/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fa/llm.txt b/docs/i18n/fa/llm.txt index b222f977adf5..c6de54f27114 100644 --- a/docs/i18n/fa/llm.txt +++ b/docs/i18n/fa/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt index ee6328847ce9..dda61789985a 100644 --- a/docs/i18n/fi/llm.txt +++ b/docs/i18n/fi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt index b981f381ff04..4dd7139b7e09 100644 --- a/docs/i18n/fr/llm.txt +++ b/docs/i18n/fr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/gu/llm.txt b/docs/i18n/gu/llm.txt index a5ffba4c86fc..ecb0741ac4f8 100644 --- a/docs/i18n/gu/llm.txt +++ b/docs/i18n/gu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt index 97b10b910bf5..de5e856bac4c 100644 --- a/docs/i18n/he/llm.txt +++ b/docs/i18n/he/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt index 13cf7a5e14c5..2a6d15b8697a 100644 --- a/docs/i18n/hi/llm.txt +++ b/docs/i18n/hi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt index 12d55db0aa44..078d59d88eee 100644 --- a/docs/i18n/hu/llm.txt +++ b/docs/i18n/hu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt index bafb1ea87667..ba42452fc49e 100644 --- a/docs/i18n/id/llm.txt +++ b/docs/i18n/id/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/in/llm.txt b/docs/i18n/in/llm.txt index 416ca93e85a7..79c9a6042651 100644 --- a/docs/i18n/in/llm.txt +++ b/docs/i18n/in/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt index 7c155d5b8a64..12f0f6555a4c 100644 --- a/docs/i18n/it/llm.txt +++ b/docs/i18n/it/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt index 8f3361f565f6..87ca4d8a177c 100644 --- a/docs/i18n/ja/llm.txt +++ b/docs/i18n/ja/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt index 882a73a3fc11..a6203810bedf 100644 --- a/docs/i18n/ko/llm.txt +++ b/docs/i18n/ko/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/mr/llm.txt b/docs/i18n/mr/llm.txt index 5a5a15aafe18..dc8c256fa970 100644 --- a/docs/i18n/mr/llm.txt +++ b/docs/i18n/mr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt index 9dca2bead653..9efa530fb7d1 100644 --- a/docs/i18n/ms/llm.txt +++ b/docs/i18n/ms/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt index 7c8a64b955de..838311afb370 100644 --- a/docs/i18n/nl/llm.txt +++ b/docs/i18n/nl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt index 00721607ea4c..2e7e4f30e4b8 100644 --- a/docs/i18n/no/llm.txt +++ b/docs/i18n/no/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt index cbda0932809b..f40f7ff96c74 100644 --- a/docs/i18n/phi/llm.txt +++ b/docs/i18n/phi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt index 8266f33a9d2a..b8e216bf4108 100644 --- a/docs/i18n/pl/llm.txt +++ b/docs/i18n/pl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt index 53dbfb7774df..92402048cc2e 100644 --- a/docs/i18n/pt-BR/llm.txt +++ b/docs/i18n/pt-BR/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt index a45d3424ffbe..1d8ea55c53bb 100644 --- a/docs/i18n/pt/llm.txt +++ b/docs/i18n/pt/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt index 696627ccd79b..cb8a15c28f70 100644 --- a/docs/i18n/ro/llm.txt +++ b/docs/i18n/ro/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt index ce054cd01e91..1a1c527403df 100644 --- a/docs/i18n/ru/llm.txt +++ b/docs/i18n/ru/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt index 099b49301263..9ecbf58571b7 100644 --- a/docs/i18n/sk/llm.txt +++ b/docs/i18n/sk/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt index f59efc12e934..e3e56731c6b9 100644 --- a/docs/i18n/sv/llm.txt +++ b/docs/i18n/sv/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sw/llm.txt b/docs/i18n/sw/llm.txt index 35f14edd40f5..9f19d4a92efd 100644 --- a/docs/i18n/sw/llm.txt +++ b/docs/i18n/sw/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ta/llm.txt b/docs/i18n/ta/llm.txt index 76627c0648e3..952ef1584c2f 100644 --- a/docs/i18n/ta/llm.txt +++ b/docs/i18n/ta/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/te/llm.txt b/docs/i18n/te/llm.txt index fe0c9aa843f5..7ca0ad8bc914 100644 --- a/docs/i18n/te/llm.txt +++ b/docs/i18n/te/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt index 5686499ba266..fa1c6d5ba90c 100644 --- a/docs/i18n/th/llm.txt +++ b/docs/i18n/th/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt index 2add1f0955b2..72ffb228ff9b 100644 --- a/docs/i18n/tr/llm.txt +++ b/docs/i18n/tr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt index 29eda2adb0d2..f2fe37d1f85e 100644 --- a/docs/i18n/uk-UA/llm.txt +++ b/docs/i18n/uk-UA/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ur/llm.txt b/docs/i18n/ur/llm.txt index 76473b3d266f..e731f3ccf575 100644 --- a/docs/i18n/ur/llm.txt +++ b/docs/i18n/ur/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt index 9d491c439d38..38e56d30c5e7 100644 --- a/docs/i18n/vi/llm.txt +++ b/docs/i18n/vi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt index 49e8eb4d4821..b3ca89b4d17c 100644 --- a/docs/i18n/zh-CN/llm.txt +++ b/docs/i18n/zh-CN/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/zh-TW/llm.txt b/docs/i18n/zh-TW/llm.txt index 861c1e655159..a9c38ef389ee 100644 --- a/docs/i18n/zh-TW/llm.txt +++ b/docs/i18n/zh-TW/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/openapi.yaml b/docs/openapi.yaml index e51b6091aeac..73941f37ead8 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -8893,12 +8893,20 @@ components: ComboCreate: type: object - required: [name, model] + required: [name, models] properties: name: type: string - model: - type: string + models: + type: array + minItems: 1 + items: + oneOf: + - type: string + description: "provider/model reference" + - type: object + description: "structured combo step (provider, model, weight, ...)" + additionalProperties: true strategy: type: string enum: @@ -8920,14 +8928,3 @@ components: - context-optimized - fusion default: priority - nodes: - type: array - items: - type: object - properties: - connectionId: - type: string - weight: - type: integer - priority: - type: integer diff --git a/docs/ops/REDIS_PRODUCTION_CONFIG.md b/docs/ops/REDIS_PRODUCTION_CONFIG.md index e8749c7bbcdb..ada9ad445672 100644 --- a/docs/ops/REDIS_PRODUCTION_CONFIG.md +++ b/docs/ops/REDIS_PRODUCTION_CONFIG.md @@ -14,9 +14,12 @@ workloads: | Workload | Driver | Client Factory | Key Pattern | |---|---|---|---| -| Rate limiting | `rateLimiter.ts` | `getRedisClient()` — lazy `ioredis` singleton | Lua‑atomic rate limit windows | -| Auth cache | `apiKeys.ts` | Reuses `rateLimiter`'s client | `auth:api_key:` with TTL | -| Quota store | `redisQuotaStore.ts` | Separate `getRedisClient(url)` singleton | Configurable per-instance | +| Rate limiting | `rateLimiter.ts` | `getRedisClient()` — lazy `ioredis` singleton | `rl:*` Lua‑atomic rate limit windows | +| Auth cache | `apiKeys.ts` | Reuses `rateLimiter`'s client | `auth:api_key:` with TTL | +| Quota store | `redisQuotaStore.ts` | Separate `getRedisClient(url)` singleton | `quota:*` configurable per-instance | + +All three workloads share one namespace prefix so OmniRoute can co-exist with other apps on a +single Redis instance (e.g. `127.0.0.1:6379`). See [Key Namespacing](#key-namespacing). --- @@ -25,6 +28,7 @@ workloads: | Setting | Value | Where | |---|---|---| | `REDIS_URL` env var | `redis://redis:6379` (compose), optional | `rateLimiter.ts:5`, `.env.example` | +| `REDIS_KEY_PREFIX` env var | `omniroute:` (default) | `rateLimiter.ts`, `redisQuotaStore.ts`, `.env.example` | | `QUOTA_STORE_REDIS_URL` env var | separate, can differ from `REDIS_URL` | `quota/storeFactory.ts` | | `QUOTA_STORE_DRIVER` | `"sqlite"` (default), `"redis"` optional | `quota/storeFactory.ts` | | ioredis `maxRetriesPerRequest` | `3` | `rateLimiter.ts` client creation | @@ -36,6 +40,29 @@ workloads: --- +## Key Namespacing + +OmniRoute shares a Redis instance with whatever else runs on the host. Without a namespace, +keys like `auth:api_key:` or `rl:*` could collide with keys from other applications +using the same Redis (this instance runs Redis on `127.0.0.1:6379` alongside other services). + +Set `REDIS_KEY_PREFIX` to a non-empty string to prefix **every** OmniRoute key: + +```bash +# .env — all OmniRoute keys become omniroute:rl:*, omniroute:auth:*, omniroute:quota:* +REDIS_KEY_PREFIX=omniroute: +``` + +- **Default:** `omniroute:` (applied when `REDIS_KEY_PREFIX` is unset or blank). +- **Applied to:** rate limiter + auth cache (shared `ioredis` client via `keyPrefix`) and the + quota store (`KEY_PREFIX = "${REDIS_KEY_PREFIX}quota"`). +- **Changing the prefix** when keys already exist in Redis orphans the old keys (they expire + via TTL / LRU). Safe to change; no migration needed. +- **ioredis `keyPrefix`** automatically prepends the prefix on writes **and** strips it on reads, + so application code never sees the prefix. + +--- + ## Recommended Production Tuning ### 1. Connection Pool / Client Options (ioredis `Redis` constructor) diff --git a/docs/reference/API_REFERENCE.md b/docs/reference/API_REFERENCE.md index 359ae4750eca..1b71551e373f 100644 --- a/docs/reference/API_REFERENCE.md +++ b/docs/reference/API_REFERENCE.md @@ -636,6 +636,49 @@ completion. --- +## Self-service usage (`/api/usage/om-usage`) + +Any API key can read **its own** usage and quotas — no management auth. This is the endpoint a +client (CLI, the OmniCopilot panel) uses to show a key holder their spend. + +```bash +# Text form (the historical contract — plain text for a terminal) +curl -H "Authorization: Bearer " \ + http://localhost:20128/api/usage/om-usage + +# Structured form — what a UI consumes +curl -H "Authorization: Bearer " \ + "http://localhost:20128/api/usage/om-usage?format=json" +``` + +The key must have **`allowUsageCommand`** enabled (off by default — the dashboard's API-key +manager toggles it per key). Without it the endpoint answers `403`. + +`?format=json` returns a discriminated shape so a caller never reads a data field off a +refusal. On success: + +```jsonc +{ + "allowed": true, + // present only when the key opted into per-key usage limits (daily/weekly USD): + "personal": { "dailySpentUsd": 1.25, "dailyLimitUsd": 5, "dailyResetAtIso": "…", "weeklySpentUsd": 8, "weeklyLimitUsd": 20, "weeklyResetAtIso": "…" /* … */ }, + // the selected provider quota snapshot, or null when nothing is cached yet: + "provider": { "connectionId": "…", "provider": "claude", "plan": "…", "quotas": { /* … */ } }, + // every connection's snapshot, so a UI can render several providers side by side: + "providers": [ { "connectionId": "…", "provider": "claude", /* … */ }, { "provider": "codex", /* … */ } ] +} +``` + +On refusal (`401` bad key / `403` not allowed) the same route returns +`{ "allowed": false, "error": { "message": "…" } }` — a present-but-empty `personal`/`provider` +(key allowed, nothing learned yet) is a different state from a refusal, and only the JSON form +distinguishes them. + +**Auth:** the caller's own Bearer API key, validated with `isValidApiKey` — this is *not* the +management surface (`/api/keys/…`), which stays behind `requireManagementAuth`. + +--- + ## Semantic Cache ```bash diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index b3c12467f8cf..ae06e6209059 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -1343,6 +1343,7 @@ Provider quota endpoints, network tunnels (Tailscale, Ngrok, MITM debug proxy), | `OMNIROUTE_REDIS_BIND_HOST` | `127.0.0.1` | `bin/cli/commands/redis.mjs` | Host interface the 1-click Redis launcher publishes on. The launcher starts Redis WITHOUT a password, so binding `0.0.0.0` hands every host on your LAN an unauthenticated Redis — only widen this if you also set a password on the instance yourself. | | `REDIS_BIND_HOST` | `127.0.0.1` | `docker-compose.yml` | Host interface docker-compose publishes the Redis sidecar on (#9286). The compose Redis runs without `requirepass`; app containers reach it over the compose network (`redis:6379`) — the published port exists only for host-side tooling. `0.0.0.0` exposes an unauthenticated Redis to the whole LAN. | | `REDIS_PORT` | `6379` | `docker-compose.yml` | Host port for the compose Redis sidecar. | +| `REDIS_KEY_PREFIX` | `omniroute:` | `src/shared/utils/rateLimiter.ts` | Namespace prefix applied to every OmniRoute Redis key (rate limiter, auth cache, quota store). Prevents key collisions when the Redis instance is shared with other apps (#11042). | | `OMNIROUTE_INTERNAL_SERVICE_TOKEN` | _(unset — mechanism disabled)_ | `src/lib/api/internalServiceAuth.ts` | Shared secret for identity-preserving internal REST hops (#9260): OmniRoute components calling other local OmniRoute routes send it as `x-omniroute-internal-service-token` so the original caller identity is preserved. Compared with `timingSafeEqual`. | | `OMNIROUTE_INTERNAL_SERVICE_TOKEN_FILE` | _(unset)_ | `src/lib/api/internalServiceAuth.ts` | Secret-file variant of the internal service token: path to a file whose trimmed content is the token. Only consulted when the inline var is unset. | | `OPENROUTER_PROVIDER_STATS_ENABLED` | `true` | `src/lib/catalog/openrouterProviderStats.ts` | Enrich the dashboard providers list with OpenRouter weekly ranking stats (#9324). On by default; set `false` to skip the background fetch entirely (non-blocking, never fatal). | diff --git a/docs/reference/FREE_TIERS.md b/docs/reference/FREE_TIERS.md index a0ff47191595..927dccf72013 100644 --- a/docs/reference/FREE_TIERS.md +++ b/docs/reference/FREE_TIERS.md @@ -120,7 +120,6 @@ A 50-agent web-research pass (official docs + last-7-days news, adversarially ve | `firecrawl` | caution | Cloud API ToS has no explicit personal-proxy prohibition found, but the open-source self-hosted version is AGPL-3.0 (re… | | `gemini` | caution | ToS explicitly states the free tier is for "developers building with Google AI models for professional or business purp… | | `groq` | caution | Services Agreement §6.3 prohibits reselling, sublicensing, or distributing API access; §3.2 bars reselling/leasing acco… | -| `hackclub` | caution | Service is explicitly scoped to Hack Club teen members building projects/learning; no public ToS found explicitly permi… | | `huggingchat` | caution | Hugging Face ToS does not explicitly ban personal self-hosted proxies, but supplemental terms (referenced but not fully… | | `huggingface` | caution | ToS grants a limited license to access/use the service; the document does not explicitly permit or forbid a single-user… | | `hyperbolic` | caution | ToS grants API access "solely for your own personal or internal business purposes" and explicitly prohibits licensing, … | @@ -222,7 +221,6 @@ A 50-agent web-research pass (official docs + last-7-days news, adversarially ve | `duckduckgo-web` | keyless | — | — | avoid | 6 | | `freemodel-dev` | keyless | — | — | unknown | 4 | | `friendliai` | keyless | — | — | avoid | 2 | -| `hackclub` | keyless | — | — | caution | 3 | | `iflytek` | keyless | — | — | avoid | 1 | | `inference-net` | keyless | — | — | caution | 3 | | `liquid` | keyless | — | — | unknown | 1 | @@ -280,7 +278,6 @@ A 50-agent web-research pass (official docs + last-7-days news, adversarially ve - **`gitlawb`** — The shipped freeNote "Free tier available" is effectively stale. The original free MiMo access was removed in May 2026; the only remaining "free" option is a temporary promotional model (Nemotron 3 U… - **`gitlawb-gmi`** — Partially still accurate — free tier exists but is now narrowed to a single model (Nemotron 3 Ultra) after MiMo free access was revoked in late May 2026. The shipped note "Free tier available" unders… - **`groq`** — The shipped freeNote "30 RPM / 14.4K RPD" is accurate only for llama-3.1-8b-instant. Most other models (including llama-3.3-70b-versatile) have a much lower 1K RPD cap. The note omits model-specific … -- **`hackclub`** — The "30+ models" count appears accurate and still matches. The core offering remains free for Hack Club members. No evidence of tightening — still "$0 ALWAYS FREE" per the homepage. The freeNote omit… - **`huggingchat`** — The shipped freeNote ("Free LLM chat — no subscription required. Rate limits apply.") is partially accurate but significantly understates the restrictions. The free tier now operates on a hard $0.10/… - **`huggingface`** — Significantly tightened. The shipped freeNote ("Free Inference API for thousands of models") implied unlimited/generous free access, but as of mid-2025 the free tier is capped at $0.10/month in recur… - **`hyperbolic`** — Our shipped freeNote says "$1-5 trial credits on signup" — the $1 trial credit portion is accurate, but the "$5" figure refers to the minimum deposit required to unlock GPU rental (not free credits g… diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index d8e32e8da43d..571fe0e904cf 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -10,7 +10,7 @@ lastUpdated: 2026-08-21 > Regenerate with: `npm run gen:provider-reference` > **Last generated:** 2026-08-21 -Total providers: **348**. See category breakdown below. +Total providers: **349**. See category breakdown below. ## Categories @@ -120,7 +120,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | | `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | -## API Key Providers (paid / paid-with-free-credits) (232) +## API Key Providers (paid / paid-with-free-credits) (233) | ID | Alias | Name | Tags | Website | Notes | |----|-------|------|------|---------|-------| @@ -242,6 +242,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. | | `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | | `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | +| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. | | `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | | `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | | `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | diff --git a/llm.txt b/llm.txt index 3deca0171a2d..ac68d5cb4e4c 100644 --- a/llm.txt +++ b/llm.txt @@ -1,6 +1,6 @@ # OmniRoute -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -14,7 +14,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -165,7 +165,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -277,7 +277,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -434,7 +434,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -475,7 +475,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/open-sse/config/constants.ts b/open-sse/config/constants.ts index c45cf542a661..d99da4725b05 100644 --- a/open-sse/config/constants.ts +++ b/open-sse/config/constants.ts @@ -171,6 +171,7 @@ export const HTTP_STATUS = { FORBIDDEN: 403, NOT_FOUND: 404, NOT_ACCEPTABLE: 406, + UNPROCESSABLE_ENTITY: 422, REQUEST_TIMEOUT: 408, GONE: 410, RATE_LIMITED: 429, @@ -263,11 +264,17 @@ export const PROVIDER_PROFILES = { circuitBreakerReset: envInt("OMNIROUTE_CIRCUIT_BREAKER_API_KEY_RESET_MS", 30000), // Provider-level circuit breaker (entire provider cooldown after repeated failures) providerFailureThreshold: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_FAILURE_THRESHOLD", 15), // Scaled for 500+ connections (was 5) - providerFailureWindowMs: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_FAILURE_WINDOW_MS", 1800000), // 30min window (was 20min) + providerFailureWindowMs: envInt( + "OMNIROUTE_PROVIDER_BREAKER_API_KEY_FAILURE_WINDOW_MS", + 1800000 + ), // 30min window (was 20min) providerCooldownMs: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_COOLDOWN_MS", 600000), // 10min cooldown when threshold reached degradationThreshold: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_DEGRADATION_THRESHOLD", 7), maxBackoffMultiplier: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_MAX_BACKOFF_MULTIPLIER", 4), - backoffEscalationCount: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_BACKOFF_ESCALATION_COUNT", 3), + backoffEscalationCount: envInt( + "OMNIROUTE_PROVIDER_BREAKER_API_KEY_BACKOFF_ESCALATION_COUNT", + 3 + ), }, // Local providers (localhost inference backends like Ollama, LM Studio, oMLX). // Not yet wired into getProviderProfile() — will be used when local provider_nodes @@ -348,6 +355,23 @@ export const STREAM_RECOVERY = { HOLDBACK_MS: 750, BUFFER_MAX_BYTES: 65536, EARLY_RETRY_MAX: 4, + /** + * Minimum character overlap `trimContinuationOverlap` must find between the + * already-emitted text and a mid-stream continuation for the continuation to be + * accepted as a real resume, rather than an unrelated restart the model produced after + * ignoring the assistant-prefill. + * + * This is a DOCUMENTED TRADE-OFF, not a solved distinction: a model that continues + * cleanly with fewer than this many echoed characters (a legitimate, even preferred, + * outcome — there was nothing to de-duplicate) is indistinguishable, from string data + * alone, from a model that silently restarted on an unrelated sentence. Both produce a + * low/zero overlap. Rejecting below this threshold trades some false-positive rejections + * of legitimate low-overlap continuations (bounded retry, then a clean close — no data + * loss beyond that retry) against not silently gluing two unrelated fragments into one + * corrupted, unrecoverable answer. It does not eliminate the residual false negative + * either (an accidental coincidence at or above this many characters is still accepted). + */ + MIN_CONTINUATION_OVERLAP_CHARS: 8, } as const; /** diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index e6ef081aa7e7..38c17abf2d9a 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -194,9 +194,6 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "groq", modelId: "openai/gpt-oss-120b", displayName: "GPT-OSS 120B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, { provider: "groq", modelId: "openai/gpt-oss-20b", displayName: "GPT-OSS 20B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, { provider: "groq", modelId: "qwen/qwen3-32b", displayName: "Qwen3 32B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, - { provider: "hackclub", modelId: "meta-llama/llama-3.3-70b-instruct", displayName: "Llama 3.3 70B", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "hackclub", tos: "caution" }, - { provider: "hackclub", modelId: "mistralai/mistral-7b-instruct", displayName: "Mistral 7B", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "hackclub", tos: "caution" }, - { provider: "hackclub", modelId: "deepseek-ai/deepseek-coder-33b", displayName: "DeepSeek Coder 33B", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "hackclub", tos: "caution" }, { provider: "huggingchat", modelId: "baidu/ERNIE-4.5-VL-424B-A47B-Base-PT", displayName: "ERNIE 4.5 VL 424B A47B Base PT", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, { provider: "huggingchat", modelId: "CohereLabs/c4ai-command-r7b-12-2024", displayName: "Command R7B 12-2024", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, { provider: "huggingchat", modelId: "CohereLabs/command-a-reasoning-08-2025", displayName: "Command A Reasoning 08-2025", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 9c1580e7aedd..8668de2c6368 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -19,17 +19,16 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ export const GLM_SHARED_MODELS = Object.freeze([ { - // GLM-5.3 (2026-08-14): one upstream id; effort is the reasoning_effort - // param (low|high|max, default max) — the -high/-low entries below are - // OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. - // Default context window not yet published by Z.ai; 1M mirrored from - // GLM-5.2 (same base model). https://z.ai/blog/glm-5.3 + // GLM-5.3 exposes low|high|max reasoning_effort (default max); -high/-low + // are OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. + // https://docs.z.ai/guides/llm/glm-5.3 id: "glm-5.3", name: "GLM 5.3", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], }, { id: "glm-5.3-high", @@ -38,6 +37,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.3-low", @@ -46,14 +46,19 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low"], }, { + // GLM-5.2 has two positive effective tiers: low/medium map to high and xhigh + // maps to max; disabling thinking remains the separate thinking toggle. + // https://docs.z.ai/guides/capabilities/thinking id: "glm-5.2", name: "GLM 5.2", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high", "max"], }, { id: "glm-5.2-high", @@ -62,6 +67,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.2-max", @@ -70,14 +76,18 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["max"], }, { + // Earlier GLM families support the thinking toggle, not reasoning_effort. + // An explicit empty list prevents generic catalog tiers from being inferred. id: "glm-5.1", name: "GLM 5.1", contextLength: 204800, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5", @@ -86,6 +96,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5-turbo", @@ -94,6 +105,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7-flash", @@ -102,6 +114,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7", @@ -110,6 +123,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.6v", @@ -118,6 +132,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -127,6 +142,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5v", @@ -135,6 +151,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -144,6 +161,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5-air", @@ -152,6 +170,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, ]); diff --git a/open-sse/config/providerErrorRules.ts b/open-sse/config/providerErrorRules.ts index c910e3fb4d91..6f00d2c5f121 100644 --- a/open-sse/config/providerErrorRules.ts +++ b/open-sse/config/providerErrorRules.ts @@ -30,21 +30,63 @@ export type ProviderErrorRule = { export type ProviderErrorRuleMatch = { reason: ConfiguredErrorReason; /** - * Intended lock scope. #10334: this field is CONSUMED end-to-end only for - * providers in `HONORS_RULE_LOCK_SCOPE_PROVIDERS` (agentrouter-exclusive - * today, gated by `honorsRuleLockScope()`) — for those, `checkFallbackError` - * surfaces it as `ruleScope` on its return value for the persistence layer - * to honor instead of re-deriving scope from `hasPerModelQuota()`. For - * every other provider it remains INFORMATIONAL: `getProviderErrorRuleMatch` - * callers still read only `reason`/`cooldownMs`, and the actual lock scope - * is decided independently by each call site. Widening the allowlist is - * tracked as a follow-up — see `docs/architecture/RESILIENCE_GUIDE.md` §7. + * Intended lock scope. #10334: for a BUILT-IN catalog rule, this field is + * CONSUMED end-to-end only for providers in `HONORS_RULE_LOCK_SCOPE_PROVIDERS` + * (agentrouter-exclusive today, gated by `honorsRuleLockScope()`) — for those, + * `checkFallbackError` surfaces it as `ruleScope` on its return value for the + * persistence layer to honor instead of re-deriving scope from + * `hasPerModelQuota()`. For every other built-in-rule provider it remains + * INFORMATIONAL. #11104: an OPERATOR-declared rule (`OperatorProviderErrorRule`) + * is exempt from this allowlist — `honorsRuleLockScope()` always returns true + * when the provider has one, since the operator already opted in by declaring + * the rule. Widening `HONORS_RULE_LOCK_SCOPE_PROVIDERS` itself (for a new + * built-in catalog rule) is tracked as a follow-up — see + * `docs/architecture/RESILIENCE_GUIDE.md` §7. */ scope: "model" | "provider" | "connection"; /** Optional explicit cooldown; falls back to the existing per-reason defaults. */ cooldownMs?: number; }; +/** + * Operator-declared per-provider error rule (settings-driven). + * + * Mirrors the catalog `ProviderErrorRule` but is data-only so an operator can + * add a scope/cooldown/reason override for a provider without editing this + * file. `match` is a plain case-insensitive SUBSTRING of the error body — never + * a RegExp — so an operator-supplied pattern can never introduce a ReDoS on the + * error-classification hot path. Bounded to <= 50 rules total by the settings + * schema. An operator rule is consulted BEFORE the built-in `providerRuleRegistry` + * and wins on the first status+substring match for a provider. + */ +export type OperatorProviderErrorRule = { + status: number; + match: string; + scope: "model" | "provider" | "connection"; + reason?: ConfiguredErrorReason; + cooldownMs?: number; +}; + +let operatorProviderErrorRules: Record = {}; + +/** + * Inject operator-declared rules. Called from the runtime-settings applier + * (`applyRuntimeSettings`) once at boot and on every settings update, with the + * value validated by the settings schema. Pass `undefined`/empty/null to clear. + * Provider keys are lowercased so lookups are case-insensitive. + */ +export function setOperatorProviderErrorRules( + rules: Record | undefined | null +): void { + operatorProviderErrorRules = {}; + if (!rules) return; + for (const [provider, list] of Object.entries(rules)) { + if (Array.isArray(list) && list.length > 0) { + operatorProviderErrorRules[provider.toLowerCase()] = list; + } + } +} + // ─── Opencode ─────────────────────────────────────────────────────────────────── // Opencode Go uses an account-wide quota. The body usually says "rate limit // reached" but the presence of `x-ratelimit-remaining-requests: 0` is the @@ -272,11 +314,21 @@ export const providerRuleRegistry = new Map([ * FULL_TEXT_RULE_PROVIDERS: that set controls what body a rule matches against * (input), this one controls whether the matched scope changes caller behavior * (output). A provider could need one without the other. + * + * Providers with an operator-declared rule (`setOperatorProviderErrorRules`) + * are honored too, without being added here: the allowlist exists to gate + * BUILT-IN catalog rules, which change default behavior for every operator + * running that provider — an operator rule is already an explicit, per-operator + * opt-in, so gating it a second time behind this list would make the settings + * mechanism (#11104) silently inert for every provider except the ones listed + * below. See `hasOperatorRuleForProvider`. */ const HONORS_RULE_LOCK_SCOPE_PROVIDERS = new Set(["agentrouter"]); export function honorsRuleLockScope(provider: string | null | undefined): boolean { - return !!provider && HONORS_RULE_LOCK_SCOPE_PROVIDERS.has(provider.toLowerCase()); + if (!provider) return false; + const key = provider.toLowerCase(); + return HONORS_RULE_LOCK_SCOPE_PROVIDERS.has(key) || hasOperatorRuleForProvider(key); } /** @@ -310,28 +362,51 @@ export function egressBucketedLockProviders(): string[] { } /** - * Providers whose rules match on the FULL upstream error text. - * checkFallbackError's rule lookup normally passes only the structured + * Providers whose BUILT-IN catalog rules match on the FULL upstream error + * text. checkFallbackError's rule lookup normally passes only the structured * error ({code, type} — message stripped by the combo callers), which is * enough for header/status/code rules but blind to body-text markers like * agentrouter's "额度不足". Providers in this set get the raw error text as * the match body instead. EXCLUSIVE allowlist by owner decision (2026-08-13): * adding a provider here is an explicit opt-in — the default path for every * other provider must remain byte-for-byte unchanged. + * + * Operator-declared rules bypass this allowlist entirely (see + * `hasOperatorRuleForProvider`): the operator's `match` is a literal substring + * of the error body by construction, so a rule that never sees body text could + * never match anything, defeating the point of declaring it. */ const FULL_TEXT_RULE_PROVIDERS = new Set(["agentrouter"]); +/** + * True when an operator has declared at least one rule for this provider via + * `settings.providerErrorRules` (injected through `setOperatorProviderErrorRules`). + * Presence of the rule IS the opt-in — no separate allowlist to maintain, and + * no widening decision needed as new operators configure new providers. + */ +export function hasOperatorRuleForProvider(provider: string | null | undefined): boolean { + if (!provider) return false; + const rules = operatorProviderErrorRules[provider.toLowerCase()]; + return !!rules && rules.length > 0; +} + /** * Resolve the body handed to getProviderErrorRuleMatch inside - * checkFallbackError: full error text for FULL_TEXT_RULE_PROVIDERS, - * the structured error for everyone else. + * checkFallbackError: full error text for FULL_TEXT_RULE_PROVIDERS or any + * provider with an operator-declared rule, the structured error for everyone + * else. */ export function resolveRuleMatchBody( provider: string | null | undefined, structuredError: unknown, errorText: string | null | undefined ): unknown { - if (provider && FULL_TEXT_RULE_PROVIDERS.has(provider.toLowerCase()) && errorText) { + if ( + provider && + (FULL_TEXT_RULE_PROVIDERS.has(provider.toLowerCase()) || + hasOperatorRuleForProvider(provider)) && + errorText + ) { return errorText; } return structuredError ?? null; @@ -346,10 +421,32 @@ export function getProviderErrorRuleMatch( provider: string | null | undefined, status: number, headers: Headers | Record | null | undefined, - body?: unknown + body?: unknown, + operatorRules?: Record ): ProviderErrorRuleMatch | null { if (!provider) return null; - const rules = providerRuleRegistry.get(provider.toLowerCase()); + const key = provider.toLowerCase(); + + // Operator-declared rules win first: an operator can override any catalog + // rule for a provider without editing this file. `operatorRules` is the + // injected source (tests / direct callers); when omitted we fall back to the + // settings-backed cache populated by `setOperatorProviderErrorRules`. + const opRules = (operatorRules ?? operatorProviderErrorRules)?.[key]; + if (opRules && opRules.length > 0) { + const text = typeof body === "string" ? body : JSON.stringify(body ?? ""); + const lowered = text.toLowerCase(); + for (const r of opRules) { + if (r.status === status && lowered.includes(r.match.toLowerCase())) { + return { + reason: r.reason ?? "quota_exhausted", + scope: r.scope, + cooldownMs: r.cooldownMs, + }; + } + } + } + + const rules = providerRuleRegistry.get(key); if (!rules) return null; // Normalize headers: accept either a `Headers` object (from `fetch()`) or // a plain record. Provider rules access headers via plain object indexing. diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts index bb3b6e0f9fe3..69841c055ce1 100644 --- a/open-sse/config/providerRegistry.ts +++ b/open-sse/config/providerRegistry.ts @@ -10,6 +10,10 @@ export { } from "./providers/registry/alibaba/index.ts"; export { REGISTRY } from "./providers/index.ts"; import { REGISTRY } from "./providers/index.ts"; +// Imported from `privateHost` rather than `outboundUrlGuard`: this module is reachable from +// `ProviderDetailPageClient.tsx`, so anything it pulls in has to survive a browser bundle +// (#11122). `privateHost` is platform-free by contract; the guard module is not. +import { isPrivateHost } from "@/shared/network/privateHost"; import { RegistryModel, REASONING_UNSUPPORTED, @@ -132,11 +136,8 @@ export function isLocalProvider(baseUrl?: string | null): boolean { try { const url = new URL(baseUrl); const hostname = url.hostname; - // Strictly matching 172.16.0.0/12 (Docker/local) and explicitly blocking ::1 per SSRF hardening - return ( - LOCAL_HOSTNAMES.has(hostname) || - /^172\.(1[6-9]|2[0-9]|3[0-1])\.\d{1,3}\.\d{1,3}$/.test(hostname) - ); + if (!hostname) return false; + return LOCAL_HOSTNAMES.has(hostname) || isPrivateHost(hostname); } catch { return false; } diff --git a/open-sse/config/providers/index.ts b/open-sse/config/providers/index.ts index 00861b24225d..3e2c812e8510 100644 --- a/open-sse/config/providers/index.ts +++ b/open-sse/config/providers/index.ts @@ -70,7 +70,6 @@ import { togetherProvider } from "./registry/together/index.ts"; import { cohereProvider } from "./registry/cohere/index.ts"; import { cursorProvider, cursor_apiProvider } from "./registry/cursor/index.ts"; import { volcengineProvider } from "./registry/volcengine/index.ts"; -import { hackclubProvider } from "./registry/hackclub/index.ts"; import { freetheaiProvider } from "./registry/freetheai/index.ts"; import { g4f_groqProvider } from "./registry/g4f-groq/index.ts"; import { g4f_geminiProvider } from "./registry/g4f-gemini/index.ts"; @@ -265,6 +264,7 @@ import { freeAiProvider } from "./registry/free-ai/index.ts"; import { voidAiProvider } from "./registry/void-ai/index.ts"; import { helixmindProvider } from "./registry/helixmind/index.ts"; import { tabitokenProvider } from "./registry/tabitoken/index.ts"; +import { logfareProvider } from "./registry/logfare/index.ts"; export const REGISTRY: Record = { aimlapi: aimlapiProvider, @@ -336,7 +336,6 @@ export const REGISTRY: Record = { cursor: cursorProvider, "cursor-api": cursor_apiProvider, volcengine: volcengineProvider, - hackclub: hackclubProvider, freetheai: freetheaiProvider, "g4f-groq": g4f_groqProvider, "g4f-gemini": g4f_geminiProvider, @@ -534,4 +533,5 @@ export const REGISTRY: Record = { "void-ai": voidAiProvider, helixmind: helixmindProvider, tabitoken: tabitokenProvider, + logfare: logfareProvider, }; diff --git a/open-sse/config/providers/registry/cline/index.ts b/open-sse/config/providers/registry/cline/index.ts index 21e8c600ed09..aecf811f193b 100644 --- a/open-sse/config/providers/registry/cline/index.ts +++ b/open-sse/config/providers/registry/cline/index.ts @@ -27,7 +27,7 @@ export const clineProvider: RegistryEntry = { // the official free bucket and text-output models advertised as zero-cost. models: [ { - id: "zai/glm-5.2", + id: "z-ai/glm-5.2", name: "GLM 5.2", toolCalling: true, supportsReasoning: true, diff --git a/open-sse/config/providers/registry/hackclub/index.ts b/open-sse/config/providers/registry/hackclub/index.ts deleted file mode 100644 index 272ee5f86caa..000000000000 --- a/open-sse/config/providers/registry/hackclub/index.ts +++ /dev/null @@ -1,19 +0,0 @@ -import type { RegistryEntry } from "../../shared.ts"; - -export const hackclubProvider: RegistryEntry = { - id: "hackclub", - alias: "hc", - format: "openai", - executor: "default", - baseUrl: "https://ai.hackclub.com/proxy/v1/chat/completions", - modelsUrl: "https://ai.hackclub.com/proxy/v1/models", - authType: "optional", - authHeader: "bearer", - passthroughModels: true, - defaultContextLength: 128000, - models: [ - { id: "meta-llama/llama-3.3-70b-instruct", name: "Llama 3.3 70B" }, - { id: "mistralai/mistral-7b-instruct", name: "Mistral 7B" }, - { id: "deepseek-ai/deepseek-coder-33b", name: "DeepSeek Coder 33B" }, - ], -}; diff --git a/open-sse/config/providers/registry/kimi/web/index.ts b/open-sse/config/providers/registry/kimi/web/index.ts index 4344194974a0..ddeea4350baa 100644 --- a/open-sse/config/providers/registry/kimi/web/index.ts +++ b/open-sse/config/providers/registry/kimi/web/index.ts @@ -12,10 +12,9 @@ export const kimi_webProvider: RegistryEntry = { alias: "kimi-web", format: "openai", executor: "kimi-web", - // International consumer chat — the legacy `kimi.moonshot.cn` domain now - // redirects every non-CN visitor to www.kimi.com, which speaks a different - // Connect-RPC API. See `open-sse/executors/kimi-web.ts` for the wire format. - baseUrl: "https://www.kimi.com", + // International consumer chat — Connect-RPC API at www.kimi.ai. + // See `open-sse/executors/kimi-web.ts` for the wire format. + baseUrl: "https://www.kimi.ai", authType: "apikey", authHeader: "Authorization", // Curated-only catalog. Agent Swarm is excluded because it requires Kimi's diff --git a/open-sse/config/providers/registry/logfare/index.ts b/open-sse/config/providers/registry/logfare/index.ts new file mode 100644 index 000000000000..9b16b5a2f4e0 --- /dev/null +++ b/open-sse/config/providers/registry/logfare/index.ts @@ -0,0 +1,25 @@ +import type { RegistryEntry } from "../../shared.ts"; +import { buildOpenAiCompatibleRegistryEntry } from "../../shared.ts"; + +/** + * Logfare — free OpenAI-compatible LLM inference provider. + * + * Live-verified 2026-08-21: GET https://logfare.ai/v1/models returns a real + * catalog (20 models; 11 chat-capable incl. kimi-k3, deepseek-v4-pro, + * glm-5.2, gpt-5.6-luna, minimax-m3). Auth is a Bearer API key issued + * instantly at https://logfare.ai/register (username/password, no email). + * + * ⚠️ Privacy: in exchange for free inference, Logfare logs every request + * (prompts, completions, metadata). After PII scrubbing this may feed their + * private internal evaluation datasets. Users can opt out at /consent; see + * https://logfare.ai/tos and https://logfare.ai/privacy. The dashboard card + * surfaces this via freeNote. + */ +export const logfareProvider: RegistryEntry = buildOpenAiCompatibleRegistryEntry({ + id: "logfare", + alias: "logfare", + baseUrl: "https://logfare.ai/v1/chat/completions", + modelsUrl: "https://logfare.ai/v1/models", + models: [], + passthroughModels: true, +}); diff --git a/open-sse/config/providers/registry/opencode/go/index.ts b/open-sse/config/providers/registry/opencode/go/index.ts index 14304426c5fd..abebd92c0fc6 100644 --- a/open-sse/config/providers/registry/opencode/go/index.ts +++ b/open-sse/config/providers/registry/opencode/go/index.ts @@ -14,8 +14,6 @@ export const opencode_goProvider: RegistryEntry = { authPrefix: "Bearer", defaultContextLength: 200000, models: [ - ...OPENCODE_ZEN_GO_SHARED_MODELS, - // Port from decolua/9router 8efacc11: align with official Go endpoints — // glm-5.2 is now advertised and Kimi chat traffic must route through // `kimi-k2.7-code` (the live API rejects the plain `kimi-k2.7` alias for @@ -26,6 +24,10 @@ export const opencode_goProvider: RegistryEntry = { { id: "glm-5.2", name: "GLM-5.2", supportsReasoning: true }, { id: "glm-5.2-high", name: "GLM-5.2 (high effort)", supportsReasoning: true }, { id: "glm-5.2-max", name: "GLM-5.2 (max effort)", supportsReasoning: true }, + + ...OPENCODE_ZEN_GO_SHARED_MODELS, + // models[0] (glm-5.2) is the dashboard default (LlmChatCard/ProviderTestSlideOver take models[0]). + { id: "glm-5.1", name: "GLM-5.1" }, { id: "glm-5", name: "GLM-5" }, // kimi-k2.7-code declared identically on opencode-zen — see OPENCODE_ZEN_GO_SHARED_MODELS. diff --git a/open-sse/config/providers/registry/opencode/zen/index.ts b/open-sse/config/providers/registry/opencode/zen/index.ts index 76b3811e39a1..9fdca2fc0a5c 100644 --- a/open-sse/config/providers/registry/opencode/zen/index.ts +++ b/open-sse/config/providers/registry/opencode/zen/index.ts @@ -16,8 +16,6 @@ export const opencode_zenProvider: RegistryEntry = { // from the live API response so new models work without a code deploy. passthroughModels: true, models: [ - ...OPENCODE_ZEN_GO_SHARED_MODELS, - // ── Chat / Coding ────────────────────────────────────────── // #2900: big-pickle's upstream runs DeepSeek thinking mode — declare the // interleaved reasoning_content contract so follow-up/tool-use turns replay @@ -28,6 +26,10 @@ export const opencode_zenProvider: RegistryEntry = { supportsReasoning: true, interleavedField: "reasoning_content", }, + + ...OPENCODE_ZEN_GO_SHARED_MODELS, + // models[0] (big-pickle) is the dashboard default; SHARED spread kept after it. + { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, { id: "gpt-5.6-luna", name: "GPT 5.6 Luna" }, @@ -67,6 +69,8 @@ export const opencode_zenProvider: RegistryEntry = { supportsReasoning: true, targetFormat: "openai-responses", }, + // Explicit wire-format overlay of the base opencode provider's muse-spark entry + // (targetFormat: openai-responses). Keep in sync with base on catalog syncs. { id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", diff --git a/open-sse/config/providers/registry/zcode/index.ts b/open-sse/config/providers/registry/zcode/index.ts index cd2a4eece64c..65e2e1c3d3d9 100644 --- a/open-sse/config/providers/registry/zcode/index.ts +++ b/open-sse/config/providers/registry/zcode/index.ts @@ -1,6 +1,17 @@ import type { RegistryEntry } from "../../shared.ts"; import { GLM_SHARED_MODELS } from "../../../glmProvider.ts"; +const GLM_EXECUTOR_EFFORT_ALIASES = new Set([ + "glm-5.3-high", + "glm-5.3-low", + "glm-5.2-high", + "glm-5.2-max", +]); + +export const ZCODE_MODELS = GLM_SHARED_MODELS.filter( + (model) => !GLM_EXECUTOR_EFFORT_ALIASES.has(model.id) +).map((model) => ({ ...model, supportedThinkingEfforts: [] })); + /** * Local ZCode app-server backend. Authentication remains in the user's local * ZCode profile (`builtin:zai-coding-plan`); OmniRoute does not receive or @@ -14,5 +25,7 @@ export const zcodeProvider: RegistryEntry = { baseUrl: "zcode://app-server/stdio", authType: "none", authHeader: "none", - models: [...GLM_SHARED_MODELS], + // ZCode's app-server transport does not consume reasoning_effort; keep thinking + // capability metadata without advertising aliases or tiers that it would ignore. + models: ZCODE_MODELS, }; diff --git a/open-sse/config/searchRegistry.ts b/open-sse/config/searchRegistry.ts index b3ab36ccd944..28136655fe29 100644 --- a/open-sse/config/searchRegistry.ts +++ b/open-sse/config/searchRegistry.ts @@ -10,6 +10,8 @@ * perplexity-search reuses credentials from the "perplexity" chat provider. */ +import { isProviderBlockedByIdOrAlias } from "@/shared/utils/noAuthProviders"; + export interface SearchProviderConfig { id: string; name: string; @@ -394,16 +396,18 @@ export function supportsSearchType( /** * Get all search providers as a flat list */ -export function getAllSearchProviders(): Array<{ +export function getAllSearchProviders(blockedProviders: string[] = []): Array<{ id: string; name: string; searchTypes: string[]; }> { - return Object.values(SEARCH_PROVIDERS).map((p) => ({ - id: p.id, - name: p.name, - searchTypes: p.searchTypes, - })); + return Object.values(SEARCH_PROVIDERS) + .filter((p) => !p.disabled && !isProviderBlockedByIdOrAlias(p.id, blockedProviders)) + .map((p) => ({ + id: p.id, + name: p.name, + searchTypes: p.searchTypes, + })); } /** diff --git a/open-sse/executors/accountRotation.ts b/open-sse/executors/accountRotation.ts index a64321e83d24..67115bdd8d03 100644 --- a/open-sse/executors/accountRotation.ts +++ b/open-sse/executors/accountRotation.ts @@ -1,7 +1,7 @@ /** * Shared multi-account rotation mechanics for noauth executors that round-robin * across several "accounts" (fingerprints), each with an optional dedicated - * proxy — currently `OpencodeExecutor` and `MimocodeExecutor`. + * proxy — currently `OpencodeExecutor`. * * Extracted after both executors independently implemented the same * pickAccount/markCooldown/markSuccess skeleton with the same exponential @@ -120,3 +120,58 @@ export function maskAccountId(fingerprint: string): string { export function isNetworkErrorRotatable(account: RotatableAccount): boolean { return account.proxy !== null; } + +/** + * Detect an *empty* upstream rejection: a 400 whose body carries no usable + * completion — the kind `OpencodeExecutor` must rotate/retry on instead of + * propagating as a fatal success. + * + * Signature is deliberately strict and scoped to the observed malformed + * envelope (`choices[0].message` with no `error`, no real `content`, + * `finish_reason: null`): + * - status must be exactly 400 (anything else → false); + * - body must parse and contain a `choices` array with at least one entry + * holding a `message` object; + * - an `error` field (present or empty) → false, so genuine 400s keep + * propagating immediately (#10460 precedent: classify by signature before + * rotating); + * - `tool_calls` / `reasoning_content` → false (real content); + * - `message.content` absent / null / "" → eligible; any other value + * (non-empty text, number, block array…) → false (conservative); + * - a literal `finish_reason` (not null) → false (a completed, if empty, turn). + * + * Does NOT reuse `detectMalformedNonStream` (diagnostics.ts): that classifier + * also flags `{error:{…}}` bodies as `empty_choices`, which would rotate on + * real errors — a false-positive class with a history here. + */ +export function isEmptyUpstreamRejection(status: number, bodyText: string): boolean { + if (status !== 400) return false; + let parsed: unknown; + try { + parsed = JSON.parse(bodyText); + } catch { + return false; + } + const choices = (parsed as { choices?: unknown })?.choices; + if (!Array.isArray(choices) || choices.length === 0) return false; + const first = choices[0] as { message?: unknown; finish_reason?: unknown }; + if (typeof first !== "object" || first === null) return false; + const rawMessage = (first as { message?: unknown }).message; + if (typeof rawMessage === "undefined" || rawMessage === null) return false; + if (typeof parsed !== "object" || parsed === null) return false; + if ("error" in (parsed as Record)) return false; + const msg = rawMessage as Record; + if ("tool_calls" in msg) return false; + if ("reasoning_content" in msg) return false; + const content = msg.content; + if (content !== undefined && content !== null && content !== "") return false; + if (first.finish_reason !== null && first.finish_reason !== undefined) return false; + return true; +} + +/** Best-effort extraction of the upstream `chatcmpl_*` id from a response body, + * for observability logging. Returns `"unknown"` when absent or unparseable. */ +export function extractChatcmplId(bodyText: string): string { + const match = /"id"\s*:\s*"(chatcmpl_[^"]+)"/.exec(bodyText); + return match ? match[1] : "unknown"; +} diff --git a/open-sse/executors/base.ts b/open-sse/executors/base.ts index 66e10b67e910..1c13442aff03 100644 --- a/open-sse/executors/base.ts +++ b/open-sse/executors/base.ts @@ -20,6 +20,10 @@ import { recordLearnedThinkingCap, parseThinkingBudgetMax, } from "../services/learnedThinkingCaps.ts"; +import { + recordLearnedReasoningEffort, + parseReasoningEffortEnum, +} from "../services/learnedReasoningEffortCaps.ts"; import { getParamFilterConfig, addParamToBlocklist, @@ -826,6 +830,9 @@ export class BaseExecutor { // loop. The learned cap is also recorded process-wide via // recordLearnedThinkingCap so future requests skip the 400 entirely. let thinkingBudgetClampedMax: number | null = null; + // Set by the reasoning_effort 4xx clamp-and-retry below — guards the same + // "fires at most once per URL" invariant as thinkingBudgetClampedMax above. + let reasoningEffortClamped = false; for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) { const requestCredentials = withForcedResponsesUpstream( @@ -1529,6 +1536,49 @@ export class BaseExecutor { } } + // Reasoning-effort enum 4xx clamp-and-retry (any provider/model without a + // declared reasoning_effort capability — custom OpenAI-compatible + // connections, or a registered provider the registry hasn't caught up + // with). Mirrors the thinking_budget clamp-and-retry above: parse the + // upstream-advertised accepted values, record them process-wide (so + // FUTURE requests clamp proactively via sanitizeReasoningEffortForProvider + // → getLearnedReasoningEffort), clamp the live transformedBody by + // re-running the sanitizer, and retry the same URL once. + if ( + (response.status === HTTP_STATUS.BAD_REQUEST || + response.status === HTTP_STATUS.UNPROCESSABLE_ENTITY) && + !reasoningEffortClamped && + transformedBody && + typeof transformedBody === "object" + ) { + const errText = await response + .clone() + .text() + .catch(() => ""); + const acceptedValues = parseReasoningEffortEnum(errText); + if (acceptedValues) { + reasoningEffortClamped = true; + const learned = recordLearnedReasoningEffort(this.provider, model, acceptedValues); + if (learned) { + transformedBody = sanitizeReasoningEffortForProvider( + transformedBody, + this.provider, + model, + log + ); + let retryBody = JSON.stringify(transformedBody); + if (usesClaudeCodeProtocol || this.provider === "claude") { + retryBody = await signRequestBody(retryBody); + } + log?.info?.( + "REASONING_SANITIZE", + `Upstream ${response.status} rejected reasoning_effort on ${url} — clamped to ${learned} and retrying (learned for ${this.provider}/${model})` + ); + response = await fetchWithStartTimeout(url, { ...fetchOptions, body: retryBody }); + } + } + } + // Generic reactive 400 field-downgrade; each field is stripped at most once. if ( response.status === HTTP_STATUS.BAD_REQUEST && diff --git a/open-sse/executors/base/reasoningEffort.ts b/open-sse/executors/base/reasoningEffort.ts index fa416dbc1a80..8dd99904fd6e 100644 --- a/open-sse/executors/base/reasoningEffort.ts +++ b/open-sse/executors/base/reasoningEffort.ts @@ -8,6 +8,10 @@ import { getProviderModel, getProviderModels, } from "../../config/providerModels.ts"; +import { + getLearnedReasoningEffort, + REASONING_EFFORT_ORDER, +} from "../../services/learnedReasoningEffortCaps.ts"; /** * Sanitize reasoning_effort for providers that don't accept all values. @@ -338,10 +342,24 @@ export function sanitizeReasoningEffortForProvider( const supportsXHigh = supportsXHighEffort(provider, modelStr); const supportsMax = supportsMaxEffortForProvider(provider, modelStr); + // Highest value we've actually seen this provider+model accept in a real + // upstream 4xx (learnedReasoningEffortCaps.ts) — takes priority over the + // static registry (which defaults to "supports everything" when there's no + // entry, e.g. custom OpenAI-compatible connections) and over the hardcoded + // "high" fallback below (which isn't always valid either). + const learnedCap = getLearnedReasoningEffort(provider, modelStr); + const learnedRank = learnedCap ? REASONING_EFFORT_ORDER.indexOf(learnedCap) : -1; // ── xhigh handling ────────────────────────────────────────────────────── // xhigh is OmniRoute-internal. Map it to the best effort the model accepts. if (effortStr === "xhigh") { + if (learnedCap && learnedRank < REASONING_EFFORT_ORDER.indexOf("xhigh")) { + log?.info?.( + "REASONING_SANITIZE", + `${provider}/${modelStr}: clamped reasoning_effort xhigh → ${learnedCap} (learned)` + ); + return writeEffortValue(b, learnedCap, c); + } if (supportsXHigh) return body; // model accepts xhigh natively if (supportsMax) { log?.info?.( @@ -366,6 +384,13 @@ export function sanitizeReasoningEffortForProvider( // upstream, and if it 400s the user gets a clear signal. This prevents // new models from being unusable for weeks until they're whitelisted (#8057). if (effortStr === "max") { + if (learnedCap && learnedRank < REASONING_EFFORT_ORDER.indexOf("max")) { + log?.info?.( + "REASONING_SANITIZE", + `${provider}/${modelStr}: clamped reasoning_effort max → ${learnedCap} (learned)` + ); + return writeEffortValue(b, learnedCap, c); + } if (supportsMax) return body; // explicitly known to accept max // A model that explicitly advertises its accepted tiers is safe to normalize. diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index e1674c30fb3f..b9f6cf113a29 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -517,10 +517,12 @@ export function filterNonstandardCodexSse(response: Response): Response { const transform = new TransformStream({ transform(chunk, controller) { buffer += decoder.decode(chunk, { stream: true }); - let sep: number; - while ((sep = buffer.indexOf("\n\n")) !== -1) { - const block = buffer.slice(0, sep + 2); - buffer = buffer.slice(sep + 2); + while (true) { + const separator = /\r?\n\r?\n/.exec(buffer); + if (!separator) break; + const blockEnd = separator.index + separator[0].length; + const block = buffer.slice(0, blockEnd); + buffer = buffer.slice(blockEnd); if (!dropBlock(block)) controller.enqueue(encoder.encode(block)); } }, @@ -1396,7 +1398,6 @@ export class CodexExecutor extends BaseExecutor { provider: "codex", preserveEncryptedReasoning: credentials?.providerSpecificData?.preserveEncryptedReasoning === true, - onIncompatibleReasoning: "drop", }); if (nativeCodexPassthrough) { diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index a222571022aa..59e187190763 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -73,7 +73,7 @@ type GlmEffortTier = { * `thinking.type=enabled` (5.3 no longer accepts thinking disabled). * * https://docs.z.ai/devpack/latest-model - * https://z.ai/blog/glm-5.3 + * https://docs.z.ai/guides/llm/glm-5.3 */ function parseGlmEffortTier(model: string): GlmEffortTier | null { switch (model) { diff --git a/open-sse/executors/hailuo-web.ts b/open-sse/executors/hailuo-web.ts index 79a023526cc7..7d1b839c2697 100644 --- a/open-sse/executors/hailuo-web.ts +++ b/open-sse/executors/hailuo-web.ts @@ -1,11 +1,11 @@ /** - * HailuoWebExecutor — Hailuo AI (MiniMax) web chat via www.hailuo.ai. + * HailuoWebExecutor — Hailuo AI (MiniMax) web chat via chat.minimax.io. * * Distinct from the paid API-key `minimax`/`minimax-cn` providers * (open-sse/config/providers/registry/minimax/) — this targets the free - * consumer chat product at hailuo.ai / chat.minimax.io. + * consumer chat product at chat.minimax.io. * - * Endpoint: POST https://www.hailuo.ai/v4/api/chat/msg? + * Endpoint: POST https://chat.minimax.io/v4/api/chat/msg? * Auth: `token` header — value read from the site's `_token` localStorage * entry, plus a per-request `yy` signature header. * Body: multipart/form-data — characterID, msgContent, chatID, searchMode. diff --git a/open-sse/executors/kimi-web.ts b/open-sse/executors/kimi-web.ts index 8f9c13c33282..f2c389822c31 100644 --- a/open-sse/executors/kimi-web.ts +++ b/open-sse/executors/kimi-web.ts @@ -1,10 +1,10 @@ /** - * KimiWebExecutor — Moonshot AI Chat via www.kimi.com (international) + * KimiWebExecutor — Moonshot AI Chat via www.kimi.ai (international) * * Routes requests through Kimi's consumer chat API on the international domain. * Originally this executor targeted `kimi.moonshot.cn` (mainland-CN consumer * chat). That domain now redirects every visitor outside CN to - * `https://www.kimi.com/`, which speaks a completely different API surface: + * `https://www.kimi.ai/`, which speaks a completely different API surface: * * - Endpoint: POST /apiv2/kimi.gateway.chat.v1.ChatService/Chat * - Protocol: Connect-RPC (unary envelope framing — 5-byte header + JSON) @@ -326,7 +326,7 @@ export class KimiWebExecutor extends BaseExecutor { if (!accessToken) { return makeErrorResult( 400, - "Missing Kimi access_token — log in at www.kimi.com and capture access_token from localStorage.", + "Missing Kimi access_token — log in at www.kimi.ai and capture access_token from localStorage.", body, CHAT_URL ); @@ -410,10 +410,7 @@ export class KimiWebExecutor extends BaseExecutor { const refreshToken = credentials?.refreshToken || credentials?.providerSpecificData?.refreshToken; if (refreshToken && typeof refreshToken === "string") { - const refreshRes = await exchangeKimiRefreshToken( - refreshToken, - getKimiWebBaseUrl() - ); + const refreshRes = await exchangeKimiRefreshToken(refreshToken, getKimiWebBaseUrl()); if (refreshRes.success && refreshRes.accessToken) { accessToken = refreshRes.accessToken; const retryHeaders = this.buildKimiHeaders(accessToken); diff --git a/open-sse/executors/opencode.ts b/open-sse/executors/opencode.ts index c00ae258a3d9..0829bfa871d3 100644 --- a/open-sse/executors/opencode.ts +++ b/open-sse/executors/opencode.ts @@ -15,6 +15,8 @@ import { markSuccess as markAccountSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, } from "./accountRotation.ts"; import { isNetworkRotationSharedEgressGuardEnabled } from "@/shared/utils/featureFlags"; @@ -253,14 +255,41 @@ export class OpencodeExecutor extends BaseExecutor { try { this.syncAccountsFromCredentials(input.credentials); + const { log } = input; const hasProxies = this.accounts.some((a) => a.proxy !== null); - // Fast path: no multi-account proxy wiring configured → original behavior. + // Fast path: no multi-account proxy wiring configured → original behavior, + // plus exactly ONE bounded retry when the upstream answers a 400 empty + // rejection (same predicate and logging as the rotation loop). Everything + // else passes untouched: this path deliberately preserves BaseExecutor's + // intra-URL 429 retries (no skipUpstreamRetry here). if (this.accounts.length === 1 && !hasProxies) { - return await super.execute(input); + const single = (await super.execute(input)) as HttpExecuteResult; + if (single.response.status === 400) { + let bodyText: string | null = null; + try { + bodyText = await single.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on direct account"); + } + if (bodyText !== null) { + if (isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on direct account (${chatcmplId}), retrying once…` + ); + return await super.execute(input); + } + log?.debug?.( + "OPENCODE", + "400 without error field, signature not matched on direct account — observing" + ); + } + } + return single; } - const { log } = input; // This loop only ever dispatches through super.execute() (the HTTP request // path), which always resolves the object-shaped arm of ExecutorExecuteResult // — the bare-Response arm belongs to web/scraping executors only (base.ts:290). @@ -277,8 +306,13 @@ export class OpencodeExecutor extends BaseExecutor { // network call, but proxied accounts (independent egress) are still // tried normally. let sharedEgressDown = false; + // Bounded extra attempts for empty upstream rejections: +1 for a single + // account (retry the same one), none for a multi-account fleet (rotation + // through the accounts is the retry). Avoids an unbounded loop on a + // persistently malformed upstream. + const emptyRejectionBudget = this.accounts.length === 1 ? 1 : 0; - for (let attempt = 0; attempt < this.accounts.length; attempt++) { + for (let attempt = 0; attempt < this.accounts.length + emptyRejectionBudget; attempt++) { const account = this.pickAccount(); const masked = maskAccountId(account.fingerprint); @@ -354,6 +388,34 @@ export class OpencodeExecutor extends BaseExecutor { continue; } + // Empty upstream rejection (malformed 400: no error field, no real + // content, finish_reason null — see isEmptyUpstreamRejection). Rotate/ + // retry instead of propagating it as a fatal success: the observed + // envelope was marking subagent sessions as failed. Read the body ONLY + // for a 400 (never a 200/streaming — that would buffer the good path); + // classify, log, and continue. Neitheries markCooldown nor markSuccess: + // the failure is upstream's, not this account's. + if (status === 400) { + let bodyText: string | null = null; + try { + bodyText = await result.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on empty rejection check"); + } + if (bodyText !== null && isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on account ${masked} (${chatcmplId}), rotating to next…` + ); + continue; + } + // A 400 carrying a real error (or non-empty content): propagate + // immediately, untouched — same as before this change. + this.markSuccess(account); + return result; + } + this.markSuccess(account); return result; } diff --git a/open-sse/executors/zcode.ts b/open-sse/executors/zcode.ts index 0841b4daa8e3..8f0a98ac1420 100644 --- a/open-sse/executors/zcode.ts +++ b/open-sse/executors/zcode.ts @@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto"; import { existsSync } from "node:fs"; import { homedir } from "node:os"; import { join, resolve } from "node:path"; -import { GLM_SHARED_MODELS } from "../config/glmProvider.ts"; +import { ZCODE_MODELS } from "../config/providers/registry/zcode/index.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorExecuteResult, type ProviderCredentials } from "./base.ts"; import { ZcodeAppServerClient, type ZcodeClientLike } from "./zcodeProtocol.ts"; import { buildErrorBody, errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; @@ -12,8 +12,8 @@ const DEFAULT_PROVIDER_ID = "builtin:zai-coding-plan"; const DEFAULT_TURN_TIMEOUT_MS = 120_000; const DEFAULT_POLL_INTERVAL_MS = 250; const TERMINAL_STATUSES = new Set(["completed", "idle", "paused", "error"]); -const ZCODE_MODEL_ALLOWLIST = new Set(GLM_SHARED_MODELS.map((model) => model.id)); -const DEFAULT_ZCODE_MODEL = GLM_SHARED_MODELS[0]?.id || "glm-5.2"; +const ZCODE_MODEL_ALLOWLIST = new Set(ZCODE_MODELS.map((model) => model.id)); +const DEFAULT_ZCODE_MODEL = ZCODE_MODELS[0]?.id || "glm-5.2"; type JsonRecord = Record; type OpenAIMsg = { role?: string; content?: unknown }; diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 8f11373b1fbf..fea8640f9d3c 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -516,7 +516,7 @@ export async function handleChatCore({ conversationId = null, modelPinned = false, skipResourcePressureGuard = false, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", managedLease = null, }) { let { provider, model, extendedContext } = modelInfo; @@ -1213,7 +1213,7 @@ export async function handleChatCore({ provider, preserveEncryptedReasoning: credentials?.providerSpecificData?.preserveEncryptedReasoning === true, - onIncompatibleReasoning: reasoningTransportFallback === "drop" ? "drop" : "reject", + onIncompatibleReasoning: reasoningTransportFallback === "skip" ? "reject" : "drop", } ); if (policy.incompatibleReasoning) { diff --git a/open-sse/handlers/rerank.ts b/open-sse/handlers/rerank.ts index 747e1ce50662..452e6f3500e0 100644 --- a/open-sse/handlers/rerank.ts +++ b/open-sse/handlers/rerank.ts @@ -199,6 +199,8 @@ export async function handleRerank({ return_documents, credentials, connectionId = null, + apiKeyId = null, + apiKeyName = null, }) { const startTime = Date.now(); if (!model) return errorResponse(400, "model is required"); @@ -267,10 +269,23 @@ export async function handleRerank({ if (!res.ok) { const errData = await res.json().catch(() => ({})); - return errorResponse( - res.status, - errData.message || errData.error?.message || `Provider returned HTTP ${res.status}` - ); + const errorMessage = + errData.message || errData.error?.message || `Provider returned HTTP ${res.status}`; + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: res.status, + model: `${providerId}/${modelId}`, + provider: providerId, + connectionId: connectionId || undefined, + duration: Date.now() - startTime, + requestBody, + responseBody: errData, + error: errorMessage, + apiKeyId: apiKeyId || undefined, + apiKeyName: apiKeyName || undefined, + }).catch(() => {}); + return errorResponse(res.status, errorMessage); } const data = await res.json(); @@ -289,10 +304,13 @@ export async function handleRerank({ status: 200, model: `${providerId}/${modelId}`, provider: providerId, + connectionId: connectionId || undefined, duration: Date.now() - startTime, tokens: { prompt_tokens: 0, completion_tokens: 0 }, - responseBody: { results_count: Array.isArray(result?.results) ? result.results.length : 0 }, - connectionId, + requestBody, + responseBody: result, + apiKeyId: apiKeyId || undefined, + apiKeyName: apiKeyName || undefined, }).catch(() => {}); const headers = new Headers({ ...CORS_HEADERS, "Content-Type": "application/json" }); diff --git a/open-sse/handlers/responseTranslator.ts b/open-sse/handlers/responseTranslator.ts index 01bdd4c14a74..43e03d919d64 100644 --- a/open-sse/handlers/responseTranslator.ts +++ b/open-sse/handlers/responseTranslator.ts @@ -10,6 +10,7 @@ import { caseInsensitiveToolNameLookup, restoreOpenAIToolNames, } from "../translator/helpers/toolCallHelper.ts"; +import { restoreClaudeToolName } from "../services/claudeCodeToolRemapper.ts"; import { extractReplayableResponsesReasoningText } from "../services/reasoningInputPolicy.ts"; import { sanitizeToolId } from "../translator/helpers/schemaCoercion.ts"; @@ -631,7 +632,7 @@ export function translateNonStreamingResponse( // Phase 3: Translate from OpenAI back to Client Source format if (sourceFormat === FORMATS.CLAUDE && sourceFormat !== targetFormat) { - return convertOpenAINonStreamingToClaude(toRecord(intermediateOpenAI)); + return convertOpenAINonStreamingToClaude(toRecord(intermediateOpenAI), toolNameMap ?? null); } // Gemini-family clients (Gemini, Antigravity): the streaming SSE path already @@ -667,8 +668,18 @@ function resolveReasoningText(messageObj: JsonRecord): string { /** * Helper to convert an OpenAI chat.completion JSON object to Claude format for non-streaming. + * + * `toolNameMap` carries request-side aliases; when it does not resolve a name, + * `restoreClaudeToolName` upgrades known Claude Code tools to their canonical + * PascalCase ("bash" → "Bash", "croncreate" → "CronCreate"). Without this, a + * non-streaming upstream JSON body (or a stream:true request the upstream + * answered with application/json) reaches Claude Code with lowercase tool_use + * names the CLI rejects as "No such tool available". */ -function convertOpenAINonStreamingToClaude(openaiResponse: JsonRecord): JsonRecord { +function convertOpenAINonStreamingToClaude( + openaiResponse: JsonRecord, + toolNameMap?: Map | null +): JsonRecord { const choices = openaiResponse.choices as unknown[] | undefined; const isChoicesArray = Array.isArray(choices); if (!isChoicesArray && openaiResponse.object !== "chat.completion") { @@ -717,7 +728,7 @@ function convertOpenAINonStreamingToClaude(openaiResponse: JsonRecord): JsonReco content.push({ type: "tool_use", id: sanitizeToolId(rawId), - name: toString(fn.name), + name: restoreClaudeToolName(toString(fn.name), toolNameMap ?? null), input: typeof fn.arguments === "string" ? JSON.parse(fn.arguments || "{}") : fn.arguments || {}, }); diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index 8b7f3cee3240..50c16fa9665d 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -106,6 +106,29 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("declares exact GLM reasoning-effort tiers across every shared GLM provider", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt"]) { + for (const model of getModelsByProviderId(provider)) { + expect(model.supportedThinkingEfforts, `${provider}/${model.id} effort tiers`).toEqual( + routedTiers.get(model.id) ?? [] + ); + } + } + + for (const model of getModelsByProviderId("zcode")) { + expect(model.supportedThinkingEfforts, `zcode/${model.id} effort tiers`).toEqual([]); + } + }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); diff --git a/open-sse/mcp-server/__tests__/mcp-runtime-blocked-provider-schema.test.ts b/open-sse/mcp-server/__tests__/mcp-runtime-blocked-provider-schema.test.ts new file mode 100644 index 000000000000..9f55c6a52084 --- /dev/null +++ b/open-sse/mcp-server/__tests__/mcp-runtime-blocked-provider-schema.test.ts @@ -0,0 +1,58 @@ +import { describe, it, expect } from "vitest"; +import { createMcpServer } from "../server"; +import { buildWebSearchInputSchema } from "../schemas/tools"; +import { getActiveSearchProviders } from "../schemas/providerEnums"; + +interface ToolWithSchema { + inputSchema: { + safeParse: (arg: unknown) => { success: boolean }; + }; +} + +describe("MCP Dynamic Runtime Schema Plumbing", () => { + it("getActiveSearchProviders excludes blocked providers dynamically by id or alias", () => { + const allProviders = getActiveSearchProviders([]); + expect(allProviders).toContain("serper-search"); + expect(allProviders).toContain("brave-search"); + + const filteredProviders = getActiveSearchProviders(["serper", "brave"]); + expect(filteredProviders).not.toContain("serper-search"); + expect(filteredProviders).not.toContain("brave-search"); + expect(filteredProviders.length).toBeGreaterThan(0); + }); + + it("buildWebSearchInputSchema excludes blocked providers from Zod enum", () => { + const fullSchema = buildWebSearchInputSchema([]); + const fullParsed = fullSchema.safeParse({ query: "test", provider: "serper-search" }); + expect(fullParsed.success).toBe(true); + + const blockedSchema = buildWebSearchInputSchema(["serper"]); + const blockedParsed = blockedSchema.safeParse({ query: "test", provider: "serper-search" }); + expect(blockedParsed.success).toBe(false); + }); + + it("createMcpServer with blockedProviders option registers dynamic tool schema", async () => { + const server = createMcpServer({ blockedProviders: ["serper", "brave"] }); + expect(server).toBeTruthy(); + + const registeredTools = ( + server as unknown as { _registeredTools: Record } + )._registeredTools; + expect(registeredTools).toBeTruthy(); + + const webSearchTool = registeredTools["omniroute_web_search"]; + expect(webSearchTool).toBeTruthy(); + + const parsedWithUnblocked = webSearchTool.inputSchema.safeParse({ + query: "test", + provider: "perplexity-search", + }); + expect(parsedWithUnblocked.success).toBe(true); + + const parsedWithBlocked = webSearchTool.inputSchema.safeParse({ + query: "test", + provider: "serper-search", + }); + expect(parsedWithBlocked.success).toBe(false); + }); +}); diff --git a/open-sse/mcp-server/schemas/providerEnums.ts b/open-sse/mcp-server/schemas/providerEnums.ts index adb8da067e79..61e4648c0823 100644 --- a/open-sse/mcp-server/schemas/providerEnums.ts +++ b/open-sse/mcp-server/schemas/providerEnums.ts @@ -1,12 +1,16 @@ import { SEARCH_PROVIDERS } from "../../config/searchRegistry"; +import { isProviderBlockedByIdOrAlias } from "../../../src/shared/utils/noAuthProviders"; /** * Dynamically generates a tuple of active search provider IDs for Zod enums. - * Filters out any providers marked as disabled in the registry. + * Filters out any providers marked as disabled or blocked in the security policy. */ -export function getActiveSearchProviders(): [string, ...string[]] { +export function getActiveSearchProviders(blockedProviders: string[] = []): [string, ...string[]] { const activeProviders = Object.values(SEARCH_PROVIDERS) - .filter((provider) => !provider.disabled) + .filter( + (provider) => + !provider.disabled && !isProviderBlockedByIdOrAlias(provider.id, blockedProviders) + ) .map((provider) => provider.id); if (activeProviders.length === 0) { diff --git a/open-sse/mcp-server/schemas/tools.ts b/open-sse/mcp-server/schemas/tools.ts index 7b8e459180e0..d8d83c17ea5d 100644 --- a/open-sse/mcp-server/schemas/tools.ts +++ b/open-sse/mcp-server/schemas/tools.ts @@ -462,25 +462,29 @@ export const listModelsCatalogTool: McpToolDefinition< }; // --- Tool 10: omniroute_web_search --- -export const webSearchInput = z.object({ - query: z - .string() - .min(1, "Query is required") - .max(500, "Query must be 500 characters or fewer") - .describe("The search query string"), - max_results: z - .number() - .int() - .min(1) - .max(20) - .default(5) - .describe("Maximum number of search results to return"), - search_type: z.enum(["web", "news"]).default("web").describe("Type of search to perform"), - provider: z - .enum(getActiveSearchProviders()) - .optional() - .describe("Specific search provider to use"), -}); +export function buildWebSearchInputSchema(blockedProviders: string[] = []) { + return z.object({ + query: z + .string() + .min(1, "Query is required") + .max(500, "Query must be 500 characters or fewer") + .describe("The search query string"), + max_results: z + .number() + .int() + .min(1) + .max(20) + .default(5) + .describe("Maximum number of search results to return"), + search_type: z.enum(["web", "news"]).default("web").describe("Type of search to perform"), + provider: z + .enum(getActiveSearchProviders(blockedProviders)) + .optional() + .describe("Specific search provider to use"), + }); +} + +export const webSearchInput = buildWebSearchInputSchema(); export const webSearchOutput = z.object({ id: z.string(), diff --git a/open-sse/mcp-server/server.ts b/open-sse/mcp-server/server.ts index e5f92a48f189..16e4eb8832eb 100644 --- a/open-sse/mcp-server/server.ts +++ b/open-sse/mcp-server/server.ts @@ -18,6 +18,7 @@ import { costReportInput, listModelsCatalogInput, webSearchInput, + buildWebSearchInputSchema, xSearchInput, webFetchInput, simulateRouteInput, @@ -720,7 +721,24 @@ async function handleWebFetch(args: { } } -export function createMcpServer(): McpServer { +export interface CreateMcpServerOptions { + blockedProviders?: string[] | (() => string[]); +} + +export function createMcpServer(options?: CreateMcpServerOptions): McpServer { + const resolveBlockedProviders = (): string[] => { + if (typeof options?.blockedProviders === "function") { + return options.blockedProviders(); + } + if (Array.isArray(options?.blockedProviders)) { + return options.blockedProviders; + } + return []; + }; + + const blockedProviders = resolveBlockedProviders(); + const dynamicWebSearchInput = buildWebSearchInputSchema(blockedProviders); + const server = new McpServer({ name: "omniroute", version: process.env.npm_package_version || "1.8.1", @@ -1044,10 +1062,14 @@ export function createMcpServer(): McpServer { { description: "Performs a web search using OmniRoute's search gateway. Supports multiple providers (Serper, Brave, Perplexity, Exa, Tavily) with automatic failover. Returns search results with titles, URLs, snippets, and position data.", - inputSchema: webSearchInput, + inputSchema: dynamicWebSearchInput, }, withScopeEnforcement("omniroute_web_search", (args) => - handleWebSearch(webSearchInput.parse(args)) + // Resolve per invocation (not the startup snapshot above) so a resolver + // function passed via CreateMcpServerOptions sees policy changes without + // a server rebuild. The advertised inputSchema stays a creation-time + // snapshot — MCP clients fetch it once at tools/list. + handleWebSearch(buildWebSearchInputSchema(resolveBlockedProviders()).parse(args)) ) ); diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index d63c58e94ce9..63407a38fcc8 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -21,7 +21,11 @@ import { honorsRuleLockScope, } from "../config/providerErrorRules.ts"; import * as rot from "./rotationConfig.ts"; -import { getPassthroughProviders, getProviderCategory } from "../config/providerRegistry.ts"; +import { + getPassthroughProviders, + getProviderCategory, + isLocalProvider, +} from "../config/providerRegistry.ts"; import { DEFAULT_RESILIENCE_SETTINGS, resolveResilienceSettings, @@ -37,7 +41,12 @@ import { type FailureKind, } from "../../src/shared/utils/classify429"; import { recordProviderSuccess as resetCooldownFailureCount } from "./providerCooldownTracker.ts"; -import { resolveProviderId } from "../../src/shared/constants/providers"; +import { + getProviderById, + resolveProviderId, + isLocalProvider as isLocalProviderId, + isSelfHostedChatProvider, +} from "../../src/shared/constants/providers"; import { resolveUseUpstream429BreakerHints } from "../../src/shared/utils/providerHints"; import { getCodexModelScope } from "../config/codexQuotaScopes.ts"; import { getQuotaScopedModelForProvider } from "./antigravityQuotaFamily.ts"; @@ -791,12 +800,20 @@ export function hasPerModelQuota( return connectionPassthroughModels; } if (!provider) return false; - if (getCanonicalLockProvider(provider) === "antigravity") return true; - if (getCanonicalLockProvider(provider) === "codex") return true; - if (provider === "gemini" || provider === "github") return true; - if (provider === "antigravity" || provider === "agy") return true; - if (getPassthroughProviders().has(provider)) return true; - if (isCompatibleProvider(provider)) return true; + const canonicalId = resolveProviderId(provider); + if (getCanonicalLockProvider(canonicalId) === "antigravity") return true; + if (getCanonicalLockProvider(canonicalId) === "codex") return true; + if (canonicalId === "gemini" || canonicalId === "github") return true; + if (canonicalId === "antigravity" || canonicalId === "agy") return true; + if (getPassthroughProviders().has(canonicalId)) return true; + // #11071: getPassthroughProviders() reads the open-sse REGISTRY. A provider can declare + // passthroughModels:true in the SHARED registry (src/shared/constants/providers/) and be + // absent from that set — 40 of them are, and they are neither local nor self-hosted, so the + // branch below never reaches them either. Without this lookup a missing-model 404 on one of + // those cools the whole connection instead of locking out the single model. + if (getProviderById(canonicalId)?.passthroughModels === true) return true; + if (isCompatibleProvider(canonicalId)) return true; + if (isLocalProviderId(canonicalId) || isSelfHostedChatProvider(canonicalId)) return true; return false; } diff --git a/open-sse/services/claudeCodeToolRemapper.ts b/open-sse/services/claudeCodeToolRemapper.ts index 15995a95003f..82e408e7c872 100644 --- a/open-sse/services/claudeCodeToolRemapper.ts +++ b/open-sse/services/claudeCodeToolRemapper.ts @@ -57,6 +57,10 @@ const TOOL_RENAME_MAP: Record = { cronlist: "CronList", taskoutput: "TaskOutput", taskstop: "TaskStop", + taskcreate: "TaskCreate", + taskupdate: "TaskUpdate", + tasklist: "TaskList", + taskget: "TaskGet", workflow: "Workflow", }; @@ -205,14 +209,22 @@ export function remapToolNamesInResponse( * Restore a tool name for Claude-format clients (#9008). * * Preference order: - * 1. Exact `_toolNameMap` hit (sanitized → original) - * 2. Case-insensitive match against map keys/values (Gemini/Antigravity may - * echo a lowercased name for a PascalCase Claude Code tool) - * 3. REVERSE_MAP TitleCase → lowercase fallback for clients with no request map - * (#7926 XML / OpenCode-style lowercase tools) + * 1. Exact `_toolNameMap` hit where the value differs from the key + * (sanitized → original request-side alias) + * 2. Canonical casing upgrade for known Claude Code tools + * (`croncreate` → `CronCreate`, `bash` → `Bash`, …) + * 3. Case-insensitive non-identity match against map keys/values + * (Gemini/Antigravity may echo a lowercased name for a PascalCase + * Claude Code tool) + * 4. Identity echo kept ONLY when no canonical upgrade exists + * 5. No-map fallbacks: REVERSE_MAP TitleCase → lowercase (#7926 XML / + * OpenCode-style lowercase tools), then the static table * - * Never apply REVERSE_MAP after a request-side original is known — that is what - * turned Claude Code's `Read`/`WebSearch` into `read`/`websearch`. + * Identity entries (key === value) never pin a known tool below its + * canonical casing. Some upstream gateways echo the very lowercase name + * they emitted into the alias channel; honouring that echo is what let a + * literal `croncreate` reach Claude Code as an unknown tool even though + * the request declared `CronCreate`. */ export function restoreClaudeToolName( rawName: string, @@ -220,27 +232,49 @@ export function restoreClaudeToolName( ): string { if (!rawName) return rawName; - const exact = toolNameMap?.get(rawName); - if (typeof exact === "string") return exact; + // Undefined when rawName already IS the canonical form — an input that + // maps to itself must keep flowing to the #7926 legacy paths below. + const lower = rawName.toLowerCase(); + const canonicalRaw = TOOL_RENAME_MAP[lower]; + const canonical = canonicalRaw && canonicalRaw !== rawName ? canonicalRaw : undefined; if (toolNameMap?.size) { - const lower = rawName.toLowerCase(); + const exact = toolNameMap.get(rawName); + if (typeof exact === "string" && (exact !== rawName || !canonical)) { + return exact; + } + + let identityMatch: string | undefined; for (const [sanitized, original] of toolNameMap.entries()) { - if (sanitized.toLowerCase() === lower || original.toLowerCase() === lower) { + if (sanitized.toLowerCase() !== lower && original.toLowerCase() !== lower) { + continue; + } + if (original !== rawName) { return original; } + identityMatch = original; + } + if (identityMatch !== undefined && !canonical) { + return identityMatch; } } + // Canonical echo is terminal: when the upstream echoes back the exact + // canonical form the request declared, keep it verbatim. The #7926 + // REVERSE_MAP fallbacks below would otherwise downcase it for routes that + // carry no _toolNameMap (Claude Code → OpenAI-style upstreams), which is + // what let a literal `croncreate` reach Claude Code even though the client + // declared `CronCreate` (live repro, PR #11085). + if (canonicalRaw === rawName) return rawName; + + if (canonical) return canonical; + // When no request toolNameMap is provided (e.g. non-Claude client): // If rawName is already TitleCase, apply REVERSE_MAP for #7926 backward compatibility (Bash → bash). if (!toolNameMap && REVERSE_MAP[rawName]) { return REVERSE_MAP[rawName]; } - const canonical = TOOL_RENAME_MAP[rawName.toLowerCase()]; - if (canonical) return canonical; - return REVERSE_MAP[rawName] ?? rawName; } diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index 7c7f0ce8a927..76ef5546aba4 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -617,7 +617,18 @@ export async function validateResponseQuality( try { json = JSON.parse(text); } catch { - if (text.startsWith("data:") || text.startsWith("event:")) return { valid: true }; + // An SSE stream body is expected for streamed upstreams. Besides `data:` and + // `event:` frames, the SSE spec also allows comment lines that begin with a + // colon (`:`), which providers use for keep-alives while the model is still + // generating — e.g. OpenRouter emits `: OPENROUTER PROCESSING` on slower / + // reasoning responses. A stream that opens with such a comment (or with + // leading whitespace/newlines) is still a valid stream, not malformed JSON, + // so trim and recognize the comment prefix before rejecting. Without this, + // otherwise-good streamed completions get failed as "not valid JSON". + const trimmed = text.trimStart(); + if (trimmed.startsWith("data:") || trimmed.startsWith("event:") || trimmed.startsWith(":")) { + return { valid: true }; + } return { valid: false, reason: "response is not valid JSON" }; } diff --git a/open-sse/services/compression/engines/ccr/index.ts b/open-sse/services/compression/engines/ccr/index.ts index d2136fc8f191..92869bfbc9b9 100644 --- a/open-sse/services/compression/engines/ccr/index.ts +++ b/open-sse/services/compression/engines/ccr/index.ts @@ -45,7 +45,7 @@ import { } from "../../../../../src/lib/db/ccrBlocks.ts"; import { createCompressionStats } from "../../stats.ts"; import { queryBlock, type CcrQuery } from "./ccrQuery.ts"; -import { injectCcrProtocolInstruction } from "./protocolInstruction.ts"; +import { callerSupportsCcrRetrieve, injectCcrProtocolInstruction } from "./protocolInstruction.ts"; import type { CompressionEngine, CompressionEngineApplyOptions, @@ -939,6 +939,30 @@ export const ccrEngine: CompressionEngine = { return { body, compressed: false, stats: null }; } + // #7746 follow-up: only callers whose tools[] proves they can reach + // omniroute_ccr_retrieve may have content replaced at all. For everyone + // else (plain OpenAI-compatible clients — the marker is an MCP-only + // contract) replacement would strand the original text behind a hash the + // model has no way to resolve. Skip the whole engine for them. The check + // is wrapped defensively: a malformed body must fail OPEN (no + // compression), never throw into the request pipeline. + let callerCanRetrieve = false; + try { + callerCanRetrieve = callerSupportsCcrRetrieve(body); + } catch (err) { + // Defensive: the helper is total, but if it ever throws we must fail + // OPEN (no compression) — and surface it so a future regression in the + // helper is visible instead of silently bypassing compression forever. + console.warn( + "[compression/ccr] callerSupportsCcrRetrieve threw; skipping compression:", + err instanceof Error ? err.message : err + ); + callerCanRetrieve = false; + } + if (!callerCanRetrieve) { + return { body, compressed: false, stats: null }; + } + const minChars = typeof stepConfig["minChars"] === "number" ? (stepConfig["minChars"] as number) diff --git a/open-sse/services/contextManager.ts b/open-sse/services/contextManager.ts index a2d678f107cd..6fe9e94c8e29 100644 --- a/open-sse/services/contextManager.ts +++ b/open-sse/services/contextManager.ts @@ -669,13 +669,35 @@ function purifyHistory(messages: Record[], targetTokens: number result = fixToolPairs(result); result = stripTrailingAssistantOrphanToolUse(result); - // Add summary of dropped messages + // Add summary of dropped messages. Merge the notice INTO the leading + // system/developer message instead of splicing a second system-role message + // mid-array: strict gateways (TokenRouter confirmed live 2026-08-22, see the + // PROVIDERS_SYSTEM_MUST_BE_FIRST list in src/lib/memory/injection.ts) reject + // any system message at index > 0 with HTTP 400 "System message must be at + // the beginning". When there is no leading system message, prepend one -- + // index 0 is accepted by every provider (same slot the old splice used when + // system[] was empty). if (keep < nonSystem.length) { const dropped = nonSystem.length - keep; - result.splice(system.length, 0, { - role: "system", - content: `[Context compressed: ${dropped} earlier messages removed to fit context window]`, - }); + const droppedNotice = `[Context compressed: ${dropped} earlier messages removed to fit context window]`; + const first = result[0]; + if (first && (first.role === "system" || first.role === "developer")) { + if (typeof first.content === "string") { + result[0] = { + ...first, + content: first.content ? `${droppedNotice}\n${first.content}` : droppedNotice, + }; + } else if (Array.isArray(first.content)) { + result[0] = { + ...first, + content: [{ type: "text", text: droppedNotice }, ...(first.content as unknown[])], + }; + } else { + result[0] = { ...first, content: droppedNotice }; + } + } else { + result.unshift({ role: "system", content: droppedNotice }); + } } return result; diff --git a/open-sse/services/learnedReasoningEffortCaps.ts b/open-sse/services/learnedReasoningEffortCaps.ts new file mode 100644 index 000000000000..b0125d868375 --- /dev/null +++ b/open-sse/services/learnedReasoningEffortCaps.ts @@ -0,0 +1,126 @@ +/** + * Learned Reasoning-Effort Caps — reactive capability memory for providers/models + * OmniRoute has no static registry entry for (custom OpenAI-compatible connections, + * or any registered provider whose registry entry carries no reasoning metadata). + * + * Same shape as `learnedThinkingCaps.ts` (thinking_budget), generalized from a + * numeric budget to an ordinal reasoning_effort scale: on a 4xx whose body + * enumerates the accepted values, `base.ts`'s executor calls + * `recordLearnedReasoningEffort`, which stores the highest recognized value in a + * module-level Map keyed "provider:model" (lowercased). Subsequent requests for + * the same provider+model read the cap via `getLearnedReasoningEffort` (consulted + * by `sanitizeReasoningEffortForProvider` in `executors/base/reasoningEffort.ts`) + * so the 4xx→retry round-trip is paid at most once per process per provider+model. + * + * In-memory only (same operator-accepted tradeoff as the thinking-budget cache): + * restart resets, the first request after a restart may re-learn at the cost of + * one upstream 4xx. + */ + +export const REASONING_EFFORT_ORDER: readonly string[] = [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", +]; + +// key: `${provider}:${model}` lowercased → highest value known to be accepted. +const learnedCaps = new Map(); + +function buildKey(provider: string | null | undefined, model: string | null | undefined): string { + const p = typeof provider === "string" ? provider.trim().toLowerCase() : ""; + const m = typeof model === "string" ? model.trim().toLowerCase() : ""; + if (!p || !m) return ""; + return `${p}:${m}`; +} + +function rankOf(value: string): number { + return REASONING_EFFORT_ORDER.indexOf(value); +} + +/** + * Return the learned cap for provider+model, or null when nothing has been + * learned yet (no upstream 4xx recorded). Keyed case-insensitively. + */ +export function getLearnedReasoningEffort( + provider: string | null | undefined, + model: string | null | undefined +): string | null { + const key = buildKey(provider, model); + if (!key) return null; + return learnedCaps.get(key) ?? null; +} + +/** + * Record that `acceptedValues` is the enum the upstream advertised for + * provider+model, and store the highest recognized value as the learned cap. + * Returns the stored value, or null when `acceptedValues` contained no token + * from `REASONING_EFFORT_ORDER` (nothing usable to learn) or the key is unusable. + * + * Always monotonically decreases: if a cap already stored ranks lower than the + * newly computed highest, the stored (lower) value wins and is returned + * unchanged. This keeps a later, laxer-looking response (or a race between + * concurrent requests) from ratcheting the cap back up. + */ +export function recordLearnedReasoningEffort( + provider: string | null | undefined, + model: string | null | undefined, + acceptedValues: string[] +): string | null { + const key = buildKey(provider, model); + if (!key) return null; + + let best: string | null = null; + let bestRank = -1; + for (const raw of acceptedValues) { + const rank = rankOf(raw); + if (rank > bestRank) { + bestRank = rank; + best = raw; + } + } + if (best === null) return null; + + const existing = learnedCaps.get(key); + if (existing !== undefined && rankOf(existing) <= bestRank) { + return existing; // already learned an equal-or-lower cap; keep it + } + learnedCaps.set(key, best); + return best; +} + +// Matches both prose shapes observed: OVH's `@ai-sdk/openai-compatible` +// deserializer ("expected one of `a`, `b`") and a generic vendor prose form +// ("Supported types are a, b, and c"). +const LIST_INTRO = /(?:expected one of|supported (?:types|values) are)[:\s]*([^.]+)/i; + +/** + * Extract the upstream-advertised accepted reasoning_effort values from a 4xx + * error body. Returns only tokens present in REASONING_EFFORT_ORDER (unknown + * tokens are dropped defensively) in the order they appeared, or null when the + * text names no recognized enum member. + */ +export function parseReasoningEffortEnum(errText: unknown): string[] | null { + if (typeof errText !== "string" || !errText) return null; + const match = LIST_INTRO.exec(errText); + if (!match) return null; + const tokens = match[1] + .split(/,|\band\b|&/i) + .map((t) => + t + .replace(/`/g, "") + .replace(/\([^)]*\)/g, "") + .trim() + .toLowerCase() + ) + .filter((t) => t.length > 0 && REASONING_EFFORT_ORDER.includes(t)); + return tokens.length > 0 ? tokens : null; +} + +/** Test-only: clear the learned-cap Map between tests. */ +export function __test_resetLearnedReasoningEffortCaps(): void { + learnedCaps.clear(); +} diff --git a/open-sse/services/reasoningInputPolicy.ts b/open-sse/services/reasoningInputPolicy.ts index 71a283a706d8..0e9e8d194dbf 100644 --- a/open-sse/services/reasoningInputPolicy.ts +++ b/open-sse/services/reasoningInputPolicy.ts @@ -1,5 +1,6 @@ import { REGISTRY } from "../config/providerRegistry.ts"; import type { ReasoningTransport } from "../config/providerRegistry.ts"; +import { isValidResponsesItemId } from "./responsesItemId.ts"; type JsonRecord = Record; @@ -36,7 +37,6 @@ export interface ReasoningInputPolicyOptions { export interface ReasoningInputPolicyResult { incompatibleReasoning: boolean; } - export function resolveReasoningTransport( provider: string | null | undefined, preserveEncryptedReasoning = false @@ -46,18 +46,6 @@ export function resolveReasoningTransport( return transport ?? (preserveEncryptedReasoning ? "opaque" : "plaintext"); } -export function createReasoningTransportIncompatibleError(): Error & { - statusCode: number; - errorType: string; -} { - const error = new Error( - "Reasoning continuation is not compatible with the selected target" - ) as Error & { statusCode: number; errorType: string }; - error.statusCode = 400; - error.errorType = "reasoning_transport_incompatible"; - return error; -} - function asRecord(value: unknown): JsonRecord | null { return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null; } @@ -100,11 +88,12 @@ function hasChatPlaintextReasoning(record: JsonRecord): boolean { /** * Returns only provider-authentic plaintext continuation state. Display summaries - * are excluded, and a record carrying opaque state is never cross-converted. + * and opaque-only records are excluded. Explicit plaintext remains independently + * portable when the same record also carries an opaque companion (#10949). */ export function extractReplayableResponsesReasoningText(value: unknown): string { const record = asRecord(value); - if (!record || record.type !== "reasoning" || hasOpaqueReasoningState(record)) return ""; + if (!record || record.type !== "reasoning") return ""; if (!Array.isArray(record.content)) return ""; return record.content @@ -279,22 +268,36 @@ function sanitizeResponsesInput( if (!hasPlaintext && !hasOpaque && (!hasDisplaySummary(next) || stripOrphanedSummaries)) { continue; } - if (!hasOpaque && typeof next.id === "string") delete next.id; + // `id` is only worth keeping on an opaque item with a valid string value — + // non-opaque items don't replay their id, and a malformed value (e.g. `null`, + // observed on opencode/zen) must not survive either way (#11108). + if (!hasOpaque || !isValidResponsesItemId(next.id)) delete next.id; + // Some upstreams (e.g. opencode/zen) omit `summary` entirely on opaque + // reasoning items instead of sending an empty array. Replaying that shape + // verbatim trips strict Responses-API validators that require the field + // to be present on every `input[]` item of type `reasoning` (#11108). + // Plaintext-only items intentionally have no `summary` key and must stay + // untouched. + if (hasOpaque && next.summary === undefined) next.summary = []; filtered.push(next); continue; } const cloned = { ...record }; - if (typeof cloned.id === "string") delete cloned.id; + // Strip `id` whenever present, valid or not: these items don't need a + // replayed server id, and a malformed one (e.g. `null`, same opencode/zen + // omission pattern as the reasoning branch above) must not survive either + // (#11108). + if (cloned.id !== undefined) delete cloned.id; filtered.push(cloned); } return filtered; } /** - * Applies one protocol-independent compatibility decision before request translation. - * Plaintext is portable by default; opaque state requires an explicit target declaration. - * Display summaries do not affect compatibility; stateless input drops orphan summaries. + * Projects reasoning continuation onto the selected target transport. + * Incompatible active state is dropped by default; combo routing may reject an + * attempt instead so it can fall through without mutating the request. */ export function applyReasoningInputPolicy( body: Record, @@ -306,14 +309,17 @@ export function applyReasoningInputPolicy( inputFormat === "responses" ? inspectResponsesReasoning(body.input) : inspectChatReasoning(body.messages); - const incompatibleReasoning = !isReasoningCompatible(inspection, transport); + const mixedState = inspection.hasPlaintext && inspection.hasOpaque; + const incompatibleReasoning = !mixedState && !isReasoningCompatible(inspection, transport); + // Mixed plaintext + opaque input (#10949) is never a rejection: it is projected + // onto the target transport by the per-item sanitizers below. - if (incompatibleReasoning && options.onIncompatibleReasoning !== "drop") { + if (incompatibleReasoning && options.onIncompatibleReasoning === "reject") { return { incompatibleReasoning: true }; } if (inputFormat === "chat") { - if (incompatibleReasoning && Array.isArray(body.messages)) { + if ((incompatibleReasoning || mixedState) && Array.isArray(body.messages)) { body.messages = dropIncompatibleChatReasoning(body.messages, transport); } return { incompatibleReasoning: false }; @@ -328,12 +334,13 @@ export function applyReasoningInputPolicy( }, ]; } - if (!Array.isArray(body.input)) return { incompatibleReasoning: false }; - body.input = sanitizeResponsesInput( - body.input, - transport, - incompatibleReasoning, - body.store === false - ); + if (Array.isArray(body.input)) { + body.input = sanitizeResponsesInput( + body.input, + transport, + incompatibleReasoning || mixedState, + body.store === false + ); + } return { incompatibleReasoning: false }; } diff --git a/open-sse/services/responsesInputSanitizer.ts b/open-sse/services/responsesInputSanitizer.ts index 5d81b787a10e..94cd99f934b1 100644 --- a/open-sse/services/responsesInputSanitizer.ts +++ b/open-sse/services/responsesInputSanitizer.ts @@ -1,3 +1,5 @@ +import { isValidResponsesItemId } from "./responsesItemId.ts"; + type JsonRecord = Record; type SanitizeResponsesInputOptions = { dropInternalAssistantMessages?: boolean; @@ -40,7 +42,12 @@ function sanitizeFunctionName(name: string): string { } function sanitizeInputItemId(record: JsonRecord): JsonRecord { - if (typeof record.id !== "string") return record; + if (record.id === undefined) return record; + if (!isValidResponsesItemId(record.id)) { + const next = { ...record }; + delete next.id; + return next; + } const type = typeof record.type === "string" ? record.type : ""; const expectedPrefix = SERVER_ITEM_ID_PREFIX_BY_TYPE[type]; diff --git a/open-sse/services/responsesItemId.ts b/open-sse/services/responsesItemId.ts new file mode 100644 index 000000000000..a57ac92e42f2 --- /dev/null +++ b/open-sse/services/responsesItemId.ts @@ -0,0 +1,7 @@ +// Shared by reasoningInputPolicy.ts and responsesInputSanitizer.ts: both strip a +// Responses-API `input[]` item's `id` field when it isn't a valid string before +// replay, so a malformed value (e.g. `null`, observed on opencode/zen) never +// survives to trip a strict upstream with "Expected 'id' to be a string." (#11108). +export function isValidResponsesItemId(id: unknown): id is string { + return typeof id === "string"; +} diff --git a/open-sse/services/streamRecovery.ts b/open-sse/services/streamRecovery.ts index a0a397fd87be..3a95e9a2a3fe 100644 --- a/open-sse/services/streamRecovery.ts +++ b/open-sse/services/streamRecovery.ts @@ -183,10 +183,33 @@ export function hasTerminalMarker(bytes: Uint8Array): boolean { export interface OpenAiSseScan { /** Concatenated assistant text seen across `choices[].delta.content`. */ text: string; + /** Concatenated reasoning trace seen across `choices[].delta.reasoning_content`. Some + * providers stream the entire answer here and leave `content` empty/null — tracked + * separately so a clean stop with reasoning-only output can still be recognized as + * "nothing usable was delivered" instead of "a normal empty turn". */ + reasoningText: string; /** True if any `choices[].delta.tool_calls` appeared — NEVER continue those. */ sawToolCall: boolean; - /** True if a terminal marker (`[DONE]` or a non-null `finish_reason`) appeared. */ + /** + * True only when `tool_calls` appeared in this scan AND its own + * `finish_reason: "tool_calls"` has NOT also appeared in the same scan — i.e. the + * call is still being streamed (arguments may be mid-flight). Once + * `finish_reason: "tool_calls"` closes it, the call is complete, not in flight: the + * client has the full arguments and a truncation past this point only drops + * trailing prose, which continuation can safely recover. + */ + sawToolCallInFlight: boolean; + /** + * True if a terminal marker for the OVERALL stream appeared: `[DONE]`, or a + * `finish_reason` other than `"tool_calls"`. A `finish_reason: "tool_calls"` ends + * that one choice but is not terminal for continuation purposes — the model turn + * (and the client-visible SSE) is still eligible to be resumed past it. + */ terminal: boolean; + /** The literal `finish_reason` string when present (e.g. "stop", "tool_calls", "length", + * "content_filter"), or `null` if none was seen. `terminal` alone is not precise enough + * to gate the reasoning-only-stop continuation — it must fire on `"stop"` only. */ + finishReason: string | null; /** True if at least one OpenAI-shaped `choices[].delta` was parsed (format gate). */ parsedOpenAi: boolean; } @@ -198,11 +221,22 @@ export interface OpenAiSseScan { */ export function scanOpenAiSseText(sse: string): OpenAiSseScan { let text = ""; + let reasoningText = ""; let sawToolCall = false; + let toolCallFinished = false; let terminal = false; + let finishReason: string | null = null; let parsedOpenAi = false; if (typeof sse !== "string" || sse.length === 0) { - return { text, sawToolCall, terminal, parsedOpenAi }; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight: false, + terminal, + finishReason, + parsedOpenAi, + }; } for (const line of sse.split("\n")) { const trimmed = line.trimStart(); @@ -227,14 +261,33 @@ export function scanOpenAiSseText(sse: string): OpenAiSseScan { parsedOpenAi = true; const content = (delta as { content?: unknown }).content; if (typeof content === "string") text += content; + const reasoning = (delta as { reasoning_content?: unknown }).reasoning_content; + if (typeof reasoning === "string") reasoningText += reasoning; const toolCalls = (delta as { tool_calls?: unknown }).tool_calls; if (Array.isArray(toolCalls) && toolCalls.length > 0) sawToolCall = true; } - const finishReason = (choice as { finish_reason?: unknown })?.finish_reason; - if (finishReason != null) terminal = true; + const rawFinishReason = (choice as { finish_reason?: unknown })?.finish_reason; + if (rawFinishReason === "tool_calls") { + // Ends this one choice, but the overall stream/turn stays continuable — + // never counts as the general terminal marker (see OpenAiSseScan.terminal). + toolCallFinished = true; + finishReason = "tool_calls"; + } else if (rawFinishReason != null) { + terminal = true; + if (typeof rawFinishReason === "string") finishReason = rawFinishReason; + } } } - return { text, sawToolCall, terminal, parsedOpenAi }; + const sawToolCallInFlight = sawToolCall && !toolCallFinished; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight, + terminal, + finishReason, + parsedOpenAi, + }; } export interface ContinuableBody { @@ -245,8 +298,10 @@ export interface ContinuableBody { /** * Build a re-request body that continues from `assistantSoFar` by appending it as an - * assistant turn. Returns null when the body has no `messages` array or the partial text - * is empty (nothing to continue from). Does not mutate the original. + * assistant turn. When `assistantSoFar` is empty (nothing usable was emitted yet — e.g. a + * clean stop that only produced reasoning), the messages are re-sent unchanged instead of + * appending an empty assistant turn: this simply re-asks for a real answer. Returns null + * only when the body has no `messages` array at all (nothing to continue from). */ export function makeContinuationBody( body: ContinuableBody, @@ -254,10 +309,13 @@ export function makeContinuationBody( ): (ContinuableBody & { messages: unknown[] }) | null { if (!body || typeof body !== "object") return null; if (!Array.isArray(body.messages) || body.messages.length === 0) return null; - if (typeof assistantSoFar !== "string" || assistantSoFar.length === 0) return null; + if (typeof assistantSoFar !== "string") return null; return { ...body, - messages: [...body.messages, { role: "assistant", content: assistantSoFar }], + messages: + assistantSoFar.length > 0 + ? [...body.messages, { role: "assistant", content: assistantSoFar }] + : [...body.messages], stream: true, }; } @@ -368,8 +426,13 @@ export function createRecoverableStream( let continuations = 0; let emittedTail = ""; // raw SSE not yet scanned (awaiting an event boundary) let emittedText = ""; // assistant text already delivered to the client + let emittedReasoningText = ""; // reasoning trace already delivered (never shown to the client, + // tracked only to distinguish "a real empty turn" from "the whole + // answer stayed in the reasoning channel") + let emittedFinishReason: string | null = null; // literal finish_reason last seen, if any let emittedTerminal = false; - let emittedToolCall = false; + let emittedToolCallInFlight = false; + let emittedSawToolCall = false; // any tool_call delta seen, complete or not let emittedParsedOpenAi = false; // Enqueue to the client and, when continuation is enabled, fold the chunk into the @@ -387,8 +450,11 @@ export function createRecoverableStream( emittedTail = emittedTail.slice(boundary + 2); const scan = scanOpenAiSseText(complete); emittedText += scan.text; + emittedReasoningText += scan.reasoningText; + if (scan.finishReason !== null) emittedFinishReason = scan.finishReason; if (scan.terminal) emittedTerminal = true; - if (scan.sawToolCall) emittedToolCall = true; + if (scan.sawToolCallInFlight) emittedToolCallInFlight = true; + if (scan.sawToolCall) emittedSawToolCall = true; if (scan.parsedOpenAi) emittedParsedOpenAi = true; }; @@ -396,15 +462,42 @@ export function createRecoverableStream( for (const chunk of holdback.flush()) emit(controller, chunk); }; - // A post-commit truncation is continuable only for a plain-text OpenAI-compatible - // stream that has not finished and has no tool call in flight. + // A post-commit truncation is continuable for a plain-text OpenAI-compatible stream that + // has no tool call in flight, AND either: + // - has not finished yet (the original #4131 truncation case), or + // - finished with a literal finish_reason of "stop" but delivered nothing usable while a + // non-empty reasoning trace shows the provider spent its whole turn "thinking" and never + // turned that into an answer (some providers put the entire response in + // reasoning_content and leave content empty). Gated on the LITERAL "stop" value, not the + // generic `terminal` flag — `terminal` also covers "length"/"content_filter"/a bare + // [DONE], which are out of scope for this specific recovery. + // + // Known consequence of the hallucinatedEmptyStop path (flagged in cross-review, accepted as + // inherent to tryContinue's existing design, not new to this fix): the original upstream's + // `finish_reason:"stop"` chunk was already forwarded to the client via `emit()`'s unconditional + // `controller.enqueue(chunk)` (streamRecovery.ts:381) BEFORE this scan ever runs — that is how + // `emittedFinishReason`/`emittedTerminal` get set in the first place. So the client sees an + // empty "stop" marker from the original turn, then — once the continuation succeeds — the real + // answer plus a SECOND `emitCleanTerminal` from `tryContinue`. This mirrors what already + // happens for the pre-existing truncation-continuation case (a truncated stream can likewise + // have partially delivered SSE framing before `tryContinue` appends more); it is not a new + // double-close of the underlying `ReadableStream` (`controller.close()` runs exactly once, + // after `tryContinue` returns). An SSE client that treats a bare `finish_reason:"stop"` as an + // unconditional end-of-turn (rather than waiting for `[DONE]`) may need updating separately — + // out of scope for this fix, which targets the observed opencode/OmniRoute pairing where the + // client kept the connection open. + const hallucinatedEmptyStop = () => + emittedFinishReason === "stop" && + !emittedSawToolCall && + emittedText.length === 0 && + emittedReasoningText.length > 0; + const canContinue = () => continueEnabled && continuations < maxContinuations && emittedParsedOpenAi && - !emittedToolCall && - !emittedTerminal && - emittedText.length > 0; + !emittedToolCallInFlight && + (emittedText.length > 0 ? !emittedTerminal : hallucinatedEmptyStop()); const emitCleanTerminal = (controller: ReadableStreamDefaultController) => { controller.enqueue( @@ -448,7 +541,24 @@ export function createRecoverableStream( } const scan = scanOpenAiSseText(raw); - const suffix = trimContinuationOverlap(emittedText, scan.text); + // A continuation whose overlap with what was already emitted falls below the documented + // threshold is treated as a suspected restart rather than a real resume — see + // STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS for the full trade-off rationale. This + // is a heuristic, not a proof: it deliberately trades some false-positive rejections of + // legitimate low-overlap continuations against never silently gluing two unrelated + // fragments into one corrupted message. + const overlapResult = trimContinuationOverlap(emittedText, scan.text); + const overlapChars = scan.text.length - overlapResult.length; + const isSuspectedRestart = + emittedText.length > 0 && + scan.text.length > 0 && + overlapChars < STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS; + if (isSuspectedRestart) { + if (await tryContinue(controller)) return true; + emitCleanTerminal(controller); + return true; + } + const suffix = overlapResult; if (suffix) { emit( controller, @@ -505,9 +615,11 @@ export function createRecoverableStream( const { done, value } = result; if (done) { if (holdback.committed) { - // Graceful end after commit: if it lacks a terminal marker it is a silent - // truncation — try to continue; otherwise (clean finish) just close. - if (!emittedTerminal && (await tryContinue(controller))) { + // Graceful end after commit: try a mid-stream continuation whenever canContinue() + // says the stream is worth continuing (silent truncation, or a clean-but-empty + // reasoning-only stop) — canContinue() is the single source of truth here, same as + // the read-error branch above. + if (await tryContinue(controller)) { runFinalize(); controller.close(); return; diff --git a/open-sse/translator/request/openai-responses.ts b/open-sse/translator/request/openai-responses.ts index b88665025242..d48dcdf18161 100644 --- a/open-sse/translator/request/openai-responses.ts +++ b/open-sse/translator/request/openai-responses.ts @@ -8,11 +8,7 @@ import { isOpenAIResponsesStoreEnabled } from "@/lib/providers/requestDefaults"; import { FORMATS } from "../formats.ts"; import { register } from "../registry.ts"; import { normalizeResponsesInputForChat } from "../../utils/responsesInputNormalization.ts"; -import { - createReasoningTransportIncompatibleError, - hasOpaqueReasoningState, - extractReplayableResponsesReasoningText, -} from "../../services/reasoningInputPolicy.ts"; +import { extractReplayableResponsesReasoningText } from "../../services/reasoningInputPolicy.ts"; import { getRegisteredProviders, requiresPlainStringContent, @@ -454,10 +450,8 @@ export function openaiResponsesToOpenAIRequest( if (itemType === "reasoning") { // Only genuine plaintext reasoning can cross into Chat reasoning_content. - // Opaque encrypted state and its display summary have no Chat replay form. - if (preserveReasoningContent && hasOpaqueReasoningState(item)) { - throw createReasoningTransportIncompatibleError(); - } + // Opaque encrypted state and its display summary have no Chat replay form, + // so opaque-only items are dropped while mixed items replay their plaintext. if (preserveReasoningContent) { const reasoning = extractReplayableResponsesReasoningText(item); if (reasoning) { diff --git a/open-sse/translator/request/openai-responses/toResponses.ts b/open-sse/translator/request/openai-responses/toResponses.ts index bee5efab1a27..988835edeace 100644 --- a/open-sse/translator/request/openai-responses/toResponses.ts +++ b/open-sse/translator/request/openai-responses/toResponses.ts @@ -201,6 +201,14 @@ export function openaiToOpenAIResponsesRequest( input.push({ type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], + // Strict Responses-API upstreams (e.g. opencode/zen) require `summary` + // on every `input[]` item of type "reasoning", plaintext or opaque — + // omitting it rejects the request with `input[N] missing required + // field summary`. This item is always freshly built from a chat + // client's plaintext reasoning, so there is no source summary to + // preserve; default to an empty array like the replay sanitizer does + // for opaque items in reasoningInputPolicy.ts (#11108). + summary: [], }); } diff --git a/open-sse/translator/response/openai-responses.ts b/open-sse/translator/response/openai-responses.ts index 0c59fac4fba6..01ad55d72f28 100644 --- a/open-sse/translator/response/openai-responses.ts +++ b/open-sse/translator/response/openai-responses.ts @@ -866,21 +866,25 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { function openaiResponsesToOpenAIResponseStream(chunk, state) { if (!chunk) { - if ( - state.currentToolCallNeedsNormalization && - state.currentToolCallArgsBuffer && - state.currentToolCallName - ) { - const toolSchema = state.toolSchemas?.get(state.currentToolCallName); - const argsToEmit = stripEmptyOptionalToolArgs( - state.currentToolCallArgsBuffer, - state.currentToolCallName, - toolSchema - ); - const argsStr = - typeof argsToEmit === "string" ? argsToEmit : JSON.stringify(argsToEmit ?? {}); - state.currentToolCallArgsBuffer = ""; - state.currentToolCallNeedsNormalization = false; + // Iterate every still-open call needing schema-aware normalization, not just a + // single one — multiple parallel calls can each be pending here if the stream + // ends before their output_item.done arrives. + const pendingNormalized: Array<{ index: number; argsStr: string }> = []; + if (state.toolCallByCallId instanceof Map) { + for (const entry of state.toolCallByCallId.values()) { + if (entry.needsNormalization && entry.argsBuffer) { + const toolSchema = state.toolSchemas?.get(entry.name); + const argsToEmit = stripEmptyOptionalToolArgs(entry.argsBuffer, entry.name, toolSchema); + pendingNormalized.push({ + index: entry.index, + argsStr: typeof argsToEmit === "string" ? argsToEmit : JSON.stringify(argsToEmit ?? {}), + }); + entry.argsBuffer = ""; + entry.needsNormalization = false; + } + } + } + if (pendingNormalized.length > 0) { state.finishReasonSent = true; state.finishReason = "tool_calls"; const common = { @@ -889,24 +893,21 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { created: state.created, model: state.model || "gpt-4", }; - return [ - { - ...common, - choices: [ - { - index: 0, - delta: { - tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsStr } }], - }, - finish_reason: null, - }, - ], - }, - { - ...common, - choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }], - }, - ]; + const chunks: Record[] = pendingNormalized.map(({ index, argsStr }) => ({ + ...common, + choices: [ + { + index: 0, + delta: { tool_calls: [{ index, function: { arguments: argsStr } }] }, + finish_reason: null, + }, + ], + })); + chunks.push({ + ...common, + choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }], + }); + return chunks; } // Flush: send final chunk with finish_reason if (!state.finishReasonSent && state.started) { @@ -952,7 +953,23 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { state.chatId = `chatcmpl-${Date.now()}`; state.created = Math.floor(Date.now() / 1000); state.toolCallIndex = 0; + // Kept for computeFinishReason (synthesizeCompletedToolCalls.ts) compatibility — + // that snapshot path mutates it directly and expects it to exist. In a turn with + // multiple parallel calls this only ever reflects the LAST one opened/closed, so + // it must never be used to identify a specific call — only as the "is at least + // one tool call in flight this turn" signal computeFinishReason needs, which + // toolCallIndex > 0 already covers on its own once any call has been added. state.currentToolCallId = null; + // Per-call state keyed by call_id (replaces the old singular + // currentToolCallId/ArgsBuffer/Name/NeedsNormalization/Deferred fields, which + // assumed only one function_call could ever be in flight at a time). + state.toolCallByCallId = new Map(); + // response.function_call_arguments.delta carries `item_id`/`output_index`, not + // `call_id` — resolve either one back to the call_id key used by + // toolCallByCallId (two independent reverse maps, since some upstreams omit + // item_id on delta events but still send output_index). + state.toolCallItemToCallId = new Map(); + state.toolCallOutputIndexToCallId = new Map(); } // Text content delta @@ -983,22 +1000,48 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { // Function call started if (eventType === "response.output_item.added" && data.item?.type === "function_call") { const item = data.item; - state.currentToolCallId = item.call_id || fallbackToolCallId(); - state.currentToolCallArgsBuffer = ""; // reset per-call arg buffer - state.currentToolCallDeferred = false; + const callId = item.call_id || fallbackToolCallId(); + // Kept for computeFinishReason (synthesizeCompletedToolCalls.ts) compatibility. + state.currentToolCallId = callId; + + const toolName = normalizeToolName(item.name); + // Assign this call's index NOW, at .added, not at .done — two calls opened before + // either closes (a genuine parallel dispatch) must never share an index. Deferred + // (still-nameless) calls are the one exception: they don't claim an index until + // .done resolves a real name, so a call that never gets one never burns a slot + // another call could have used. + let index: number | null = null; + if (toolName) { + index = state.toolCallIndex ?? 0; + state.toolCallIndex = index + 1; + } + + if (!(state.toolCallByCallId instanceof Map)) state.toolCallByCallId = new Map(); + state.toolCallByCallId.set(callId, { + index, + name: toolName, + argsBuffer: "", + deferred: !toolName, + needsNormalization: toolName === "Agent", + }); + if (!(state.toolCallItemToCallId instanceof Map)) state.toolCallItemToCallId = new Map(); + if (item.id) state.toolCallItemToCallId.set(item.id, callId); + // `output_index` is a top-level field on every Responses API streamed event + // (response.output_item.added/.done AND function_call_arguments.delta alike) — + // an identifier independent of item_id, for upstreams that omit item_id on delta + // events. + if (!(state.toolCallOutputIndexToCallId instanceof Map)) { + state.toolCallOutputIndexToCallId = new Map(); + } + if (data.output_index != null) state.toolCallOutputIndexToCallId.set(data.output_index, callId); // Track this call_id so response.completed doesn't synthesize a duplicate if (!state.toolCallIdsSeen) state.toolCallIdsSeen = new Set(); - if (state.currentToolCallId) state.toolCallIdsSeen.add(state.currentToolCallId); + state.toolCallIdsSeen.add(callId); - const toolName = normalizeToolName(item.name); - state.currentToolName = toolName; // track for schema lookup at done time - state.currentToolCallName = toolName; - state.currentToolCallNeedsNormalization = toolName === "Agent"; if (!toolName) { // Some Responses providers briefly emit placeholder/empty tool names. // Defer emission until output_item.done in case the final name is populated there. - state.currentToolCallDeferred = true; return null; } @@ -1013,8 +1056,8 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { delta: { tool_calls: [ { - index: state.toolCallIndex, - id: state.currentToolCallId, + index, + id: callId, type: "function", function: { name: toolName, @@ -1037,11 +1080,26 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { const argsDelta = data.delta || ""; if (!argsDelta) return null; - state.currentToolCallArgsBuffer = (state.currentToolCallArgsBuffer || "") + argsDelta; - if (state.currentToolCallDeferred || state.currentToolCallNeedsNormalization) return null; + // Resolve which in-flight call this delta belongs to. Try item_id first (the + // field the Responses API documents for this event), then output_index (also a + // top-level field on this event, and independent of item_id — covers upstreams + // that omit item_id on delta events but still send output_index). Only once both + // identifying fields are absent/unresolved do we fall back to guessing (the + // single open call, or the most recently opened one as a last resort). + const map = state.toolCallByCallId instanceof Map ? state.toolCallByCallId : null; + let callId = data.item_id ? state.toolCallItemToCallId?.get(data.item_id) : undefined; + if (!callId && data.output_index != null) { + callId = state.toolCallOutputIndexToCallId?.get(data.output_index); + } + if (!callId && map) { + callId = map.size === 1 ? [...map.keys()][0] : state.currentToolCallId; + } + const entry = callId ? map?.get(callId) : undefined; + if (!entry) return null; // #9168: buffer arguments until output_item.done for schema-aware null normalization // Previously emitted raw null values for optional enum fields (e.g. isolation: null). + entry.argsBuffer = (entry.argsBuffer || "") + argsDelta; return null; } @@ -1061,13 +1119,30 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { // carry the complete arguments only in output_item.done (no preceding delta events). if (eventType === "response.output_item.done" && data.item?.type === "function_call") { const item = data.item; - const buffered = state.currentToolCallArgsBuffer || ""; - const currentIndex = state.toolCallIndex; // capture before increment - const callId = item.call_id || state.currentToolCallId || fallbackToolCallId(); + const map = state.toolCallByCallId instanceof Map ? state.toolCallByCallId : null; + let callId = item.call_id; + if (!callId && item.id) callId = state.toolCallItemToCallId?.get(item.id); + if (!callId) callId = state.currentToolCallId || fallbackToolCallId(); + const trackedEntry = callId ? map?.get(callId) : undefined; + // Some upstreams (e.g. Codex) send the complete payload only in output_item.done, + // with no preceding output_item.added at all — there is no tracked entry to read an + // index from. + const entry = trackedEntry || { index: null, argsBuffer: "", deferred: false }; + + const buffered = entry.argsBuffer || ""; const toolName = normalizeToolName(item.name); + + // Claim (and advance) this call's index now if it wasn't assigned at .added — either + // a deferred call whose name has just now resolved, or a Codex-style done-only + // payload that never had an .added at all. A deferred call whose name is STILL empty + // never claims an index (nothing was ever emitted for it either way). + if (entry.index == null && toolName) { + entry.index = state.toolCallIndex ?? 0; + state.toolCallIndex = entry.index + 1; + } + const currentIndex = entry.index; const toolSchema = state.toolSchemas?.get(toolName); const shouldNormalizeArguments = toolName === "Agent"; - state.currentToolCallNeedsNormalization = shouldNormalizeArguments; if (toolName && state.toolCalls instanceof Map) { const completedArguments = @@ -1077,6 +1152,9 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { toolName, toolSchema ); + // Keyed by index, not insertion order — readers that need call order for + // parallel calls closed out of order should sort by this key rather than + // relying on Map iteration order. state.toolCalls.set(currentIndex, { id: callId, index: currentIndex, @@ -1095,17 +1173,17 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { if (!state.toolCallIdsSeen) state.toolCallIdsSeen = new Set(); if (callId) state.toolCallIdsSeen.add(callId); - if (state.currentToolCallDeferred) { - state.currentToolCallDeferred = false; - state.currentToolCallArgsBuffer = ""; - state.currentToolCallId = null; + // This call is fully closed — remove it from the in-flight map (bounds the map + // to genuinely in-flight calls, and keeps the single-open-call fallback in the + // function_call_arguments.delta handler correct for whichever call opens next). + if (map && callId) map.delete(callId); + if (state.currentToolCallId === callId) state.currentToolCallId = null; + if (entry.deferred) { if (!toolName) { return null; } - state.toolCallIndex++; - const terminalArguments = typeof item.arguments === "string" ? item.arguments.length > 0 @@ -1148,12 +1226,7 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { }; } - state.toolCallIndex++; - state.currentToolCallArgsBuffer = ""; // reset for next tool call - state.currentToolCallId = null; - const needsNormalization = state.currentToolCallNeedsNormalization === true; - state.currentToolCallNeedsNormalization = false; - state.currentToolCallName = ""; + const needsNormalization = shouldNormalizeArguments; // Nullable omission sentinels must be normalized before any argument bytes reach the client. // Other tool calls retain immediate argument streaming. diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index 4d8993c7670e..9adb5dbf2f6e 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -683,13 +683,15 @@ export function createDisconnectAwareStream(transformStream, streamController) { if (clientTerminalSeen) return; terminalTail += terminalDecoder.decode(chunk, { stream: true }); - if (terminalTail.length > 4096) { - terminalTail = terminalTail.slice(-4096); - } + // Scan before bounding retained state: a compaction terminal frame can + // exceed the tail budget because encrypted_content is carried inline. clientTerminalSeen = hasClientTerminalSseMarker( terminalTail, streamController.clientResponseFormat ); + if (terminalTail.length > 4096) { + terminalTail = terminalTail.slice(-4096); + } if (clientTerminalSeen) { streamController.markClientTerminalSeen?.(); } diff --git a/open-sse/utils/streamPayloadCollector.ts b/open-sse/utils/streamPayloadCollector.ts index ea1b2e658e0a..31b9e818f46c 100644 --- a/open-sse/utils/streamPayloadCollector.ts +++ b/open-sse/utils/streamPayloadCollector.ts @@ -337,6 +337,7 @@ function createOpenAIReducer(fallbackModel?: string | null): SummaryReducer { // same-name tool_calls) into its own separate tool_calls entries. const finalToolCalls: ToolCall[] = []; let nextIndex = 0; + // Normalize tool_call indexes to contiguous 0-based (OpenAI contract). for (const tc of mergedToolCalls) { const splitArgs = splitConcatenatedToolCallArguments(tc.function.arguments); if (!splitArgs) { diff --git a/open-sse/utils/streamReadiness.ts b/open-sse/utils/streamReadiness.ts index 2d06659b809b..1696a7a5c525 100644 --- a/open-sse/utils/streamReadiness.ts +++ b/open-sse/utils/streamReadiness.ts @@ -34,6 +34,14 @@ function hasUsefulValue(value: unknown): boolean { if (Array.isArray(value)) return value.some(hasUsefulValue); if (!isRecord(value)) return false; + // A Responses compaction item IS the turn's output: remote compaction + // completes with output = [{type:"compaction", encrypted_content}] and no + // assistant text. Deliberately NOT a blanket encrypted_content key — an + // encrypted reasoning item alone is not user-visible output and must keep + // tripping the #8649 empty-content guard. + // This shape is specific to Responses streams; chat-completion frames do not produce it. + if (value.type === "compaction" && hasNonEmptyString(value.encrypted_content)) return true; + for (const key of [ "content", "text", diff --git a/open-sse/utils/syncedEffortVariants.ts b/open-sse/utils/syncedEffortVariants.ts index 2a2c7f0d68c2..33c5ec8c56ba 100644 --- a/open-sse/utils/syncedEffortVariants.ts +++ b/open-sse/utils/syncedEffortVariants.ts @@ -19,17 +19,17 @@ * only when the base model's own `supportedThinkingEfforts` actually declares that tier — * never a blind string match. * - * Skipped entirely for `codex` and `kimi`-owned models: both already own a conflicting - * native `-{effort}` suffix mechanism (`splitCodexReasoningSuffix` / - * `getKimiCodeStaticThinkingPolicy`), so double-registering here would collide with their - * own alias resolution. Also skipped for any model whose id already ends in a token that - * matches a canonical effort value, to avoid colliding with a model that legitimately ends - * in an effort-like token (e.g. a model literally named "...-high"). + * Skipped entirely for `codex`, `kimi`-owned, and GLM (`glm`, `glm-cn`, `glmt`) models: + * they already own conflicting `-{effort}` aliases (`splitCodexReasoningSuffix`, + * `getKimiCodeStaticThinkingPolicy`, or `GlmExecutor::parseGlmEffortTier`), so generating + * another layer here would create invalid nested ids. Also skipped for any model whose id + * already ends in a token that matches a canonical effort value, to avoid colliding with a + * model that legitimately ends in an effort-like token (e.g. a model named "...-high"). */ import { CANONICAL_EFFORT_VALUES } from "@/shared/reasoning/effortStandardization.ts"; -/** Provider ids that already own a native `-{effort}` suffix mechanism — never double-register. */ -export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex"]); +/** Provider ids with dedicated `-{effort}` aliases — never synthesize another suffix layer. */ +export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex", "glm", "glm-cn", "glmt"]); /** Provider-id prefixes covering that mechanism's multiple connection variants (kimi-coding, kimi-coding-apikey). */ const SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES = ["kimi"]; diff --git a/package-lock.json b/package-lock.json index 57f810910e73..bf77bab602a7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -18,7 +18,6 @@ "@dnd-kit/core": "^6.3.1", "@dnd-kit/sortable": "^10.0.0", "@dnd-kit/utilities": "^3.2.2", - "@huggingface/transformers": "^4.2.0", "@lobehub/icons": "^5.16.0", "@modelcontextprotocol/sdk": "^1.29.0", "@monaco-editor/react": "^4.7.0", @@ -61,7 +60,6 @@ "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", - "onnxruntime-node": "1.24.3", "open": "^11.0.1", "ora": "^9.4.1", "parse5": "^8.0.1", @@ -156,9 +154,11 @@ }, "optionalDependencies": { "@atjsh/llmlingua-2": "3.0.0", + "@huggingface/transformers": "^4.2.0", "better-sqlite3": "^13.0.2", "js-tiktoken": "^1.0.20", "keytar": "^7.9.0", + "onnxruntime-node": "1.24.3", "sqlite-vec": "^0.1.9", "tls-client-node": "^0.2.0", "wreq-js": "^3.0.0" @@ -4510,6 +4510,7 @@ "resolved": "https://registry.npmjs.org/@huggingface/jinja/-/jinja-0.5.9.tgz", "integrity": "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw==", "license": "MIT", + "optional": true, "engines": { "node": ">=18" } @@ -4518,13 +4519,15 @@ "version": "0.1.3", "resolved": "https://registry.npmjs.org/@huggingface/tokenizers/-/tokenizers-0.1.3.tgz", "integrity": "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA==", - "license": "Apache-2.0" + "license": "Apache-2.0", + "optional": true }, "node_modules/@huggingface/transformers": { "version": "4.2.0", "resolved": "https://registry.npmjs.org/@huggingface/transformers/-/transformers-4.2.0.tgz", "integrity": "sha512-8BRCoBMH0XsWaEIamuR0LrJGAfftgHAfb2Vrffy0VKlSAE/MnUJ5/h/zTfEP3fDIft+nk7TqB8xXEyABGitBjQ==", "license": "Apache-2.0", + "optional": true, "dependencies": { "@huggingface/jinja": "^0.5.6", "@huggingface/tokenizers": "^0.1.3", @@ -9483,30 +9486,35 @@ "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/aspromise/-/aspromise-1.1.2.tgz", "integrity": "sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/base64": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/base64/-/base64-1.1.2.tgz", "integrity": "sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/codegen": { "version": "2.0.5", "resolved": "https://registry.npmjs.org/@protobufjs/codegen/-/codegen-2.0.5.tgz", "integrity": "sha512-zgXFLzW3Ap33e6d0Wlj4MGIm6Ce8O89n/apUaGNB/jx+hw+ruWEp7EwGUshdLKVRCxZW12fp9r40E1mQrf/34g==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/eventemitter": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz", "integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/fetch": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz", "integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==", + "devOptional": true, "license": "BSD-3-Clause", "dependencies": { "@protobufjs/aspromise": "^1.1.1" @@ -9516,24 +9524,28 @@ "version": "1.0.2", "resolved": "https://registry.npmjs.org/@protobufjs/float/-/float-1.0.2.tgz", "integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/path": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz", "integrity": "sha512-6JOcJ5Tm08dOHAbdR3GrvP+yUUfkjG5ePsHYczMFLq3ZmMkAD98cDgcT2iA1lJ9NVwFd4tH/iSSoe44YWkltEA==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/pool": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/@protobufjs/pool/-/pool-1.1.0.tgz", "integrity": "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/utf8": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/@protobufjs/utf8/-/utf8-1.1.1.tgz", "integrity": "sha512-oOAWABowe8EAbMyWKM0tYDKi8Yaox52D+HWZhAIJqQXbqe0xI/GV7FhLWqlEKreMkfDjshR5FKgi3mnle0h6Eg==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@radix-ui/number": { @@ -12737,6 +12749,7 @@ "version": "26.2.0", "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", + "devOptional": true, "license": "MIT", "dependencies": { "undici-types": "~8.3.0" @@ -13998,6 +14011,7 @@ "resolved": "https://registry.npmjs.org/adm-zip/-/adm-zip-0.6.0.tgz", "integrity": "sha512-XleryMhbuksdKtofnWZ9Sk+4CUTbms4Mb/EU32SZwToAyZ5RgVos/ki8n+yr0LWHOGKuakbXTuuYNHLQjhddgg==", "license": "MIT", + "optional": true, "engines": { "node": ">=14.0" } @@ -14971,7 +14985,8 @@ "resolved": "https://registry.npmjs.org/boolean/-/boolean-3.2.0.tgz", "integrity": "sha512-d0II/GO9uf9lfUHH2BQsjxzRJZBdsjgsBiW4BvhWk/3qoKwQFjIDVN19PfX8F2D/r9PCMTtLWjYVCFrpeYUzsw==", "deprecated": "Package no longer supported. Contact Support at https://www.npmjs.com/support for more info.", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/bottleneck": { "version": "2.19.5", @@ -17935,6 +17950,7 @@ "version": "1.1.4", "resolved": "https://registry.npmjs.org/define-data-property/-/define-data-property-1.1.4.tgz", "integrity": "sha512-rBMvIzlpA8v6E+SJZoo++HAYqsLrkg7MSfIinMPFhmkorw7X+dOXVJQs+QT69zGkzMyfDnIMN2Wid1+NbL3T+A==", + "devOptional": true, "license": "MIT", "dependencies": { "es-define-property": "^1.0.0", @@ -17964,6 +17980,7 @@ "version": "1.2.1", "resolved": "https://registry.npmjs.org/define-properties/-/define-properties-1.2.1.tgz", "integrity": "sha512-8QmQKqEASLd5nx0U1B1okLElbUuuttJ/AnYmRXbbbGDWh6uS208EjD4Xqq/I9wK7u0v6O08XhTWnt5XtEbR6Dg==", + "devOptional": true, "license": "MIT", "dependencies": { "define-data-property": "^1.0.1", @@ -18064,7 +18081,8 @@ "version": "2.1.0", "resolved": "https://registry.npmjs.org/detect-node/-/detect-node-2.1.0.tgz", "integrity": "sha512-T0NIuQpnTvFDATNuHN5roPwSBG83rFsuO+MXXH9/3N1eFbn4wcPjttvjMLEPWJ0RGUYgQE7cGgS3tNxbqCGM7g==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/detect-node-es": { "version": "1.1.0", @@ -18908,7 +18926,8 @@ "version": "4.1.1", "resolved": "https://registry.npmjs.org/es6-error/-/es6-error-4.1.1.tgz", "integrity": "sha512-Um/+FxMr9CISWh0bi5Zv0iOD+4cFh5qLeks1qhAopKVAJw3drgKbKySikp7wGhDL0HPeaja0P5ULZrxLkniUVg==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/es6-promisify": { "version": "7.0.0", @@ -20400,7 +20419,8 @@ "version": "25.9.23", "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-25.9.23.tgz", "integrity": "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ==", - "license": "Apache-2.0" + "license": "Apache-2.0", + "optional": true }, "node_modules/flatted": { "version": "3.4.2", @@ -21226,6 +21246,7 @@ "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-3.0.0.tgz", "integrity": "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q==", "license": "BSD-3-Clause", + "optional": true, "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", @@ -21243,6 +21264,7 @@ "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", "license": "ISC", + "optional": true, "bin": { "semver": "bin/semver.js" }, @@ -21291,6 +21313,7 @@ "version": "1.0.4", "resolved": "https://registry.npmjs.org/globalthis/-/globalthis-1.0.4.tgz", "integrity": "sha512-DpLKbNU4WylpxJykQujfCcwYWiV/Jhm50Goo0wrVILAv5jOr9d+H+UR3PhSCD2rCCEIg0uc+G+muBTwD54JhDQ==", + "devOptional": true, "license": "MIT", "dependencies": { "define-properties": "^1.2.1", @@ -21645,7 +21668,8 @@ "version": "1.0.9", "resolved": "https://registry.npmjs.org/guid-typescript/-/guid-typescript-1.0.9.tgz", "integrity": "sha512-Y8T4vYhEfwJOTbouREvG+3XDsjr8E3kIr7uf+JZ0BYloFsttiHU0WfvANVsR7TxNUJa/WpCnw/Ino/p+DeBhBQ==", - "license": "ISC" + "license": "ISC", + "optional": true }, "node_modules/hachure-fill": { "version": "0.5.2", @@ -21679,6 +21703,7 @@ "version": "1.0.2", "resolved": "https://registry.npmjs.org/has-property-descriptors/-/has-property-descriptors-1.0.2.tgz", "integrity": "sha512-55JNKuIW+vq4Ke1BjOTjM2YctQIvCT7GFzHwmfZPGo5wnrgkid0YQtnAleFSqumZm4az3n2BS+erby5ipJdgrg==", + "devOptional": true, "license": "MIT", "dependencies": { "es-define-property": "^1.0.0" @@ -24790,7 +24815,8 @@ "version": "5.0.1", "resolved": "https://registry.npmjs.org/json-stringify-safe/-/json-stringify-safe-5.0.1.tgz", "integrity": "sha512-ZClg6AaYvamvYEE82d3Iyd3vSSIjQ+odgjaTzRuO3s7toCdFKczob2i0zCh7JE8kWn17yvAWhUVxvqGwUalsRA==", - "license": "ISC" + "license": "ISC", + "optional": true }, "node_modules/json5": { "version": "2.2.3", @@ -26654,6 +26680,7 @@ "resolved": "https://registry.npmjs.org/matcher/-/matcher-3.0.0.tgz", "integrity": "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng==", "license": "MIT", + "optional": true, "dependencies": { "escape-string-regexp": "^4.0.0" }, @@ -29425,6 +29452,7 @@ "version": "1.1.1", "resolved": "https://registry.npmjs.org/object-keys/-/object-keys-1.1.1.tgz", "integrity": "sha512-NuAESUOUMrlIXOfHKzD6bpPu3tYt3xvjNdRIQ+FeT0lNb4K8WR70CaDxhuNguS2XG+GjkyMwOzsN5ZktImfhLA==", + "devOptional": true, "license": "MIT", "engines": { "node": ">= 0.4" @@ -29632,7 +29660,8 @@ "version": "1.24.3", "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.3.tgz", "integrity": "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/onnxruntime-node": { "version": "1.24.3", @@ -29640,6 +29669,7 @@ "integrity": "sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==", "hasInstallScript": true, "license": "MIT", + "optional": true, "os": [ "win32", "darwin", @@ -29656,6 +29686,7 @@ "resolved": "https://registry.npmjs.org/onnxruntime-web/-/onnxruntime-web-1.26.0-dev.20260416-b7804b056c.tgz", "integrity": "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw==", "license": "MIT", + "optional": true, "dependencies": { "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", @@ -29669,13 +29700,15 @@ "version": "5.3.2", "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", - "license": "Apache-2.0" + "license": "Apache-2.0", + "optional": true }, "node_modules/onnxruntime-web/node_modules/onnxruntime-common": { "version": "1.24.0-dev.20251116-b39e144322", "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.0-dev.20251116-b39e144322.tgz", "integrity": "sha512-BOoomdHYmNRL5r4iQ4bMvsl2t0/hzVQ3OM3PHD0gxeXu1PmggqBv3puZicEUVOA3AtHHYmqZtjMj9FOfGrATTw==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/open": { "version": "11.0.1", @@ -30906,7 +30939,8 @@ "version": "1.3.6", "resolved": "https://registry.npmjs.org/platform/-/platform-1.3.6.tgz", "integrity": "sha512-fnWVljUchTro6RiCFvCXBbNhJc2NijN7oIQxbwsyL0buWJPG85v81ehlHI9fXrJsMNgTofEoWIQeClKpgxFLrg==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/playwright": { "version": "1.62.1", @@ -31861,6 +31895,7 @@ "version": "7.6.5", "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.5.tgz", "integrity": "sha512-/FPD0nUc9jH6rfFjji9IBqOz4pcSE3CsT1m7Ep6Mdb0LxSUMj8hgl6GomOvZzpNpAqqGaXA0P3VSrZLFzIhQrw==", + "devOptional": true, "hasInstallScript": true, "license": "BSD-3-Clause", "dependencies": { @@ -31884,6 +31919,7 @@ "version": "5.3.2", "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", + "devOptional": true, "license": "Apache-2.0" }, "node_modules/proxy-addr": { @@ -33280,6 +33316,7 @@ "resolved": "https://registry.npmjs.org/roarr/-/roarr-2.15.4.tgz", "integrity": "sha512-CHhPh+UNHD2GTXNYhPWLnU8ONHdI+5DI+4EYIAOaiD63rHeYlZvyh8P+in5999TTSFgUYuKUAjzRI4mdh/p+2A==", "license": "BSD-3-Clause", + "optional": true, "dependencies": { "boolean": "^3.0.1", "detect-node": "^2.0.4", @@ -33655,7 +33692,8 @@ "version": "1.0.0", "resolved": "https://registry.npmjs.org/semver-compare/-/semver-compare-1.0.0.tgz", "integrity": "sha512-YM3/ITh2MJ5MtzaM429anh+x2jiLVjqILF4m4oyQB18W7Ggea7BfqdH/wGMK7dDiMghv/6WG7znWMwUDzJiXow==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/send": { "version": "1.2.1", @@ -33688,6 +33726,7 @@ "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-7.0.1.tgz", "integrity": "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw==", "license": "MIT", + "optional": true, "dependencies": { "type-fest": "^0.13.1" }, @@ -33703,6 +33742,7 @@ "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.13.1.tgz", "integrity": "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg==", "license": "(MIT OR CC0-1.0)", + "optional": true, "engines": { "node": ">=10" }, @@ -34474,7 +34514,8 @@ "version": "1.1.3", "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.1.3.tgz", "integrity": "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA==", - "license": "BSD-3-Clause" + "license": "BSD-3-Clause", + "optional": true }, "node_modules/sql.js": { "version": "1.14.2", @@ -36187,6 +36228,7 @@ "version": "8.3.0", "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz", "integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==", + "devOptional": true, "license": "MIT" }, "node_modules/unicode-emoji-modifier-base": { diff --git a/package.json b/package.json index bf2740f233a8..526c60cbb781 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "omniroute", "version": "3.8.50", - "description": "Unified AI router with 348 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", + "description": "Unified AI router with 349 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { "omniroute": "bin/omniroute.mjs", @@ -265,7 +265,6 @@ "@dnd-kit/core": "^6.3.1", "@dnd-kit/sortable": "^10.0.0", "@dnd-kit/utilities": "^3.2.2", - "@huggingface/transformers": "^4.2.0", "@lobehub/icons": "^5.16.0", "@modelcontextprotocol/sdk": "^1.29.0", "@monaco-editor/react": "^4.7.0", @@ -308,7 +307,6 @@ "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", - "onnxruntime-node": "1.24.3", "open": "^11.0.1", "ora": "^9.4.1", "parse5": "^8.0.1", @@ -343,9 +341,11 @@ }, "optionalDependencies": { "@atjsh/llmlingua-2": "3.0.0", + "@huggingface/transformers": "^4.2.0", "better-sqlite3": "^13.0.2", "js-tiktoken": "^1.0.20", "keytar": "^7.9.0", + "onnxruntime-node": "1.24.3", "sqlite-vec": "^0.1.9", "tls-client-node": "^0.2.0", "wreq-js": "^3.0.0" diff --git a/promise-pillars.svg b/promise-pillars.svg new file mode 100644 index 000000000000..aebefefabbfd --- /dev/null +++ b/promise-pillars.svg @@ -0,0 +1,139 @@ + + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. + + + + + + + + + + + + + + + + + + THE PROMISE + + + + One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. + + + + + + + + + + + + + + + + Never hit limits + Auto-fallback across 349 providers in + milliseconds. Quota out? The next provider + takes over — zero downtime. + + + + + + + + + + + + + + + Save up to 95% tokens + RTK + Caveman stacked compression cuts + 15–95% of eligible tokens — ~89% average + on tool-heavy sessions. + + + + + + + + + + + + + + $0 to start + 90+ providers with a free tier, 56 free + forever — Qoder, Pollinations, Cloudflare, + SiliconFlow… No card needed. + + + + + + + + + + + + + + + Every tool works + 33 coding agents — Claude Code, Codex, + Cursor, Cline, Copilot, Antigravity — + through one config. + + + + + + + + + + + + + + One endpoint + OpenAI ↔ Claude ↔ Gemini ↔ Responses API + translation. Point any tool at /v1 — + it just works. + + + + + + + + + + + + + + Production-grade + Circuit breakers, TLS stealth, MCP (110 + tools), A2A, memory, guardrails, evals — + 25,000+ tests. + + + + + + $ npm i -g omniroute  ·  point your tool at http://localhost:20128/v1  ·  $0 + MIT · OPEN SOURCE + + diff --git a/public/providers/logfare.png b/public/providers/logfare.png new file mode 100644 index 000000000000..223f6e39cdc5 Binary files /dev/null and b/public/providers/logfare.png differ diff --git a/skills/omni-webhooks/SKILL.md b/skills/omni-webhooks/SKILL.md index 251b5f58176c..c60df46ecabe 100644 --- a/skills/omni-webhooks/SKILL.md +++ b/skills/omni-webhooks/SKILL.md @@ -1,12 +1,12 @@ --- name: omni-webhooks -description: Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries. +description: Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries. --- ## Overview -Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries. +Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries. ## Authentication diff --git a/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx new file mode 100644 index 000000000000..77cf19329833 --- /dev/null +++ b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx @@ -0,0 +1,108 @@ +"use client"; + +import { useSyncExternalStore } from "react"; +import { useTranslations } from "next-intl"; +import ProviderIcon from "@/shared/components/ProviderIcon"; + +// Branded short link through our own link.omniroute.online shortener, so the +// click lands in our Kutt metrics. Points at cheaperinference.com?utm_source=omniroute +// (the URL in README.md's Open Source Friends section). Keep in sync with the +// `cheaper` slug on the shortener. +const CHEAPER_INFERENCE_URL = "https://link.omniroute.online/cheaper"; + +// Cheaper Inference brand green (#31f889). White text on it fails contrast, so +// the CTA pairs it with the dark ink from the provider's color token (colors.ts: +// cheaperinference.text = #04170d). Hex values stay in sync with that token. + +const DISMISS_STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +// Same-tab signal for the dismiss button, since writing localStorage doesn't +// fire a "storage" event in the tab that wrote it. +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; + +function isNotDismissed(): boolean { + try { + return !localStorage.getItem(DISMISS_STORAGE_KEY); + } catch { + return true; + } +} + +function subscribe(callback: () => void) { + window.addEventListener(DISMISS_EVENT, callback); + return () => window.removeEventListener(DISMISS_EVENT, callback); +} + +// SSR has no localStorage, so the server always renders the banner visible; +// useSyncExternalStore reconciles that against the real client-side value +// right after hydration, mirroring KimiSponsorBanner's pattern. +function getServerSnapshot() { + return true; +} + +/** + * Dismissable banner announcing the Cheaper Inference OmniRoute partnership on + * the dashboard home page — same size/shape as KimiSponsorBanner, no version + * gate (durable partnership, not a time-boxed offer). The logomark reuses + * . + */ +export default function CheaperInferenceSponsorBanner() { + const t = useTranslations("cheaperInferenceSponsorBanner"); + const visible = useSyncExternalStore(subscribe, isNotDismissed, getServerSnapshot); + + if (!visible) { + return null; + } + + const dismiss = () => { + try { + localStorage.setItem(DISMISS_STORAGE_KEY, "true"); + } catch { + // ignore — worst case the banner reappears next visit + } + window.dispatchEvent(new Event(DISMISS_EVENT)); + }; + + return ( +
+
+
+ +
+
+

{t("title")}

+

{t("description")}

+
+
+ +
+
+ + {t("cta")} + + + {t("partnerLinkNote")} +
+ +
+
+ ); +} diff --git a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx index 0b23976531cd..1715f17c8394 100644 --- a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx +++ b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx @@ -6,7 +6,9 @@ import { useTranslations } from "next-intl"; // Marketplace listing is the primary CTA; Open VSX (Cursor/Windsurf/VSCodium/etc.) // is called out via secondaryNote instead of a second button, to keep this banner // the same size as KimiSponsorBanner. -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +// Branded short link through our own link.omniroute.online shortener (the `vsx` +// slug), so the click lands in our Kutt metrics. +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; const DISMISS_STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; // Same-tab signal for the dismiss button, since writing localStorage doesn't diff --git a/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx b/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx index 7970c40a8eef..8d8c5c55738f 100644 --- a/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx +++ b/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx @@ -291,7 +291,9 @@ function ComboAutopilotPanel({ report }: { report: ComboAutopilotReport }) { icon="monitor_heart" label={t("comboHealthIssues")} value={report.summary.issueCount.toLocaleString()} - subValue={t("comboHealthActionable", { count: report.summary.actionableCount })} + subValue={t("comboHealthActionable", { + count: report.summary.suggestionCount ?? report.summary.actionableCount ?? 0, + })} /> - {config.reasoningTransportFallback !== "skip" && ( -

- {getI18nOrFallback( - t, - "reasoningTransportFallbackDropWarning", - "May lose continuation context or cause tool-call continuations to fail." - )} -

- )}
setFormData({ ...formData, apiKey })} /> )} - {!isNoAuthWebSessionCredential && ( -
- setFormData({ ...formData, apiKey: e.target.value })} - onKeyDown={(e) => { - if (e.key === "Enter" && !validating && !saving) { - e.preventDefault(); - handleValidate(); - } - }} - className="flex-1" - placeholder={apiCredentialPlaceholder} - hint={apiCredentialHint} - autoComplete="off" - spellCheck={false} - autoCapitalize="off" - /> -
- + {!isNoAuthWebSessionCredential && (() => { + const isCheckDisabled = + (!isCompatible && !apiKeyOptional && !formData.apiKey) || + (isGooglePse && !formData.cx.trim()) || + validating || + saving; + return ( +
+ setFormData({ ...formData, apiKey: e.target.value })} + onKeyDown={(e) => { + if (e.key === "Enter" && !isCheckDisabled) { + e.preventDefault(); + handleValidate(); + } + }} + className="flex-1" + placeholder={apiCredentialPlaceholder} + hint={apiCredentialHint} + autoComplete="off" + spellCheck={false} + autoCapitalize="off" + /> +
+ +
-
- )} + ); + })()} {isChatGptWebCodex && (
diff --git a/src/app/(dashboard)/home/page.tsx b/src/app/(dashboard)/home/page.tsx index beccde22610e..bc10df88f424 100644 --- a/src/app/(dashboard)/home/page.tsx +++ b/src/app/(dashboard)/home/page.tsx @@ -4,6 +4,7 @@ import { getSettings } from "@/lib/localDb"; import HomePageClient from "../dashboard/HomePageClient"; import BootstrapBanner from "../dashboard/BootstrapBanner"; import KimiSponsorBanner from "../dashboard/KimiSponsorBanner"; +import CheaperInferenceSponsorBanner from "../dashboard/CheaperInferenceSponsorBanner"; import VscodeCopilotBanner from "../dashboard/VscodeCopilotBanner"; import NewsBanner from "../dashboard/NewsBanner"; @@ -20,6 +21,7 @@ export default async function HomePage() { <> {isBootstrapped && } + diff --git a/src/app/api/local/redis/status/route.ts b/src/app/api/local/redis/status/route.ts index 7f7cb47bc57d..0ab37e76fdc0 100644 --- a/src/app/api/local/redis/status/route.ts +++ b/src/app/api/local/redis/status/route.ts @@ -63,21 +63,51 @@ async function pingRedis(port: string): Promise { }); } +function parseRedisUrl(url?: string): { host: string; port: number } | null { + if (!url) return null; + try { + const u = new URL(url); + return { host: u.hostname || "127.0.0.1", port: Number(u.port) || 6379 }; + } catch { + return null; + } +} + export async function GET() { const guard = isLocalRequestAllowed(); if (!guard.allowed) { - return NextResponse.json({ error: guard.reason }, { status: 403 }); + const reason = (guard as { reason?: string }).reason ?? "Forbidden: not a loopback request"; + return NextResponse.json({ error: reason }, { status: 403 }); } + // Docker/Podman container state (the 1-click launcher path). const runtime = await detectRuntime(); - if (!runtime) { - return NextResponse.json( - { exists: false, running: false, reachable: false, error: "No container runtime (podman or docker) found on PATH" }, - { status: 503 } - ); + let container = { exists: false, running: false, reachable: false }; + if (runtime) { + const { exists, running } = await containerState(runtime); + const reachable = running ? await pingRedis(HOST_PORT) : false; + container = { exists, running, reachable }; } - const { exists, running } = await containerState(runtime); - const reachable = running ? await pingRedis(HOST_PORT) : false; - return NextResponse.json({ runtime, name: CONTAINER_NAME, port: HOST_PORT, exists, running, reachable }); + // Native Redis via REDIS_URL (the production path this instance uses). OmniRoute + // is "connected" whenever REDIS_URL is configured AND the server answers — even + // when no Docker container is present. + const redisUrl = process.env.REDIS_URL?.trim() || ""; + const parsed = parseRedisUrl(redisUrl); + const redisUrlReachable = parsed ? await pingRedis(String(parsed.port)) : false; + + const running = container.running || redisUrlReachable; + const reachable = container.reachable || redisUrlReachable; + const exists = container.exists || redisUrlReachable; + + return NextResponse.json({ + runtime: runtime ?? null, + name: CONTAINER_NAME, + port: HOST_PORT, + exists, + running, + reachable, + redisUrlConfigured: Boolean(redisUrl), + redisUrlReachable, + }); } \ No newline at end of file diff --git a/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts b/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts index 62988798d77e..0939b59b554d 100644 --- a/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts +++ b/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts @@ -87,6 +87,22 @@ export function parseAlibabaModelStudioModelsForConnection( export function parseQwenCloudTextModels(data: any): any[] { return parseCuratedDashscopeModels(data, QWEN_CLOUD_TEXT_MODELS, QWEN_CLOUD_TEXT_MODEL_IDS); } + +// Perplexity's /v1/models lists the Agent API catalog (vendor-prefixed ids like +// "anthropic/claude-fable-5"), but chat requests always go to the classic +// /chat/completions endpoint, which only accepts the Sonar family. Filter +// discovery to Sonar-family ids so agent-style ids never surface as routable +// chat models (#11060). Bounded pattern — no ReDoS-prone quantifiers. +export function parsePerplexitySonarModels(data: any): any[] { + const models = Array.isArray(data?.data) + ? data.data + : Array.isArray(data?.models) + ? data.models + : []; + return models.filter( + (model: any) => typeof model?.id === "string" && /^sonar(-|$)/.test(model.id) + ); +} type ProviderModelsHeaderContext = { authType?: string; providerSpecificData?: unknown; @@ -659,6 +675,17 @@ export const PROVIDER_MODELS_CONFIG: Record = headers: { Accept: "application/json" }, parseResponse: parseClinepassRecommendedModels, }, + // Perplexity's /v1/models lists the Agent API catalog (vendor-prefixed agent + // ids), but chat only accepts the Sonar family on /chat/completions. Import + // must keep Sonar-family ids only (#11060). + perplexity: { + url: "https://api.perplexity.ai/v1/models", + method: "GET", + headers: { "Content-Type": "application/json" }, + authHeader: "Authorization", + authPrefix: "Bearer ", + parseResponse: parsePerplexitySonarModels, + }, cohere: { url: "https://api.cohere.com/v2/models", method: "GET", diff --git a/src/app/api/providers/[id]/models/discovery/providerSets.ts b/src/app/api/providers/[id]/models/discovery/providerSets.ts index 823868147904..2b35e54b01e3 100644 --- a/src/app/api/providers/[id]/models/discovery/providerSets.ts +++ b/src/app/api/providers/[id]/models/discovery/providerSets.ts @@ -95,6 +95,11 @@ export const NAMED_OPENAI_STYLE_PROVIDERS = new Set([ "internlm", "ant-ling", "nanogpt", + // Logfare (https://logfare.ai) — free OpenAI-compatible gateway live-verified + // 2026-08-21: GET https://logfare.ai/v1/models returns a real 20-model catalog + // (11 chat-capable). Live fetch keeps it fresh; the registry seed stays as the + // offline fallback. + "logfare", ]); export function isNamedOpenAIStyleProvider(provider: string): boolean { diff --git a/src/app/api/providers/[id]/route.ts b/src/app/api/providers/[id]/route.ts index 562dad074448..4f588571b87c 100644 --- a/src/app/api/providers/[id]/route.ts +++ b/src/app/api/providers/[id]/route.ts @@ -121,7 +121,17 @@ export async function PUT(request: Request, { params }: { params: Promise<{ id: const { id } = await params; const validation = validateBody(updateProviderConnectionSchema, rawBody); if (isValidationFailure(validation)) { - return NextResponse.json({ error: validation.error }, { status: 400 }); + // never drop an operator's intent silently. Surface the rejected + // keys (field paths and unrecognized-key names) alongside the existing + // error envelope so clients and the UI can tell exactly what was refused. + const rejected = [ + ...validation.error.details.map((d) => d.field).filter(Boolean), + ...validation.error.details.flatMap((d) => d.keys ?? []), + ]; + return NextResponse.json( + { error: { ...validation.error, rejected } }, + { status: 400 } + ); } const body = validation.data; const { diff --git a/src/app/api/usage/call-logs/route.ts b/src/app/api/usage/call-logs/route.ts index c340514c235a..c91a0561a5b5 100644 --- a/src/app/api/usage/call-logs/route.ts +++ b/src/app/api/usage/call-logs/route.ts @@ -3,7 +3,7 @@ export const dynamic = "force-dynamic"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { getCallLogs } from "@/lib/usageDb"; import { getCompletedDetails, getPendingById } from "@/lib/usage/usageHistory"; -import { getProviderConnections } from "@/lib/localDb"; +import { getProviderConnections } from "@/lib/db/providers"; import { getProviderNodes } from "@/models"; import { matchesSearch } from "@/shared/utils/turkishText"; @@ -27,6 +27,66 @@ function rowPriority(row: any): number { return 2; } +/** + * Applies the active filter predicates to a single merged call-log row. + * + * `getCallLogs()` already filters the persisted DB rows server-side, but the + * in-memory entries (active/pending + recently-completed) are merged in by + * `buildCallLogListRows()` and would otherwise bypass every filter except + * `correlationId`. Running the same predicates over the merged rows closes that + * gap. It is idempotent for DB rows (they already satisfy the predicate) while + * correctly excluding in-memory rows that do not match. + */ +export function rowMatchesFilter(row: any, filter: Record): boolean { + if (!filter) return true; + + if (filter.status === "error") { + if (!(Number(row?.status) >= 400 || Boolean(row?.error))) return false; + } else if (filter.status === "ok") { + if (!(Number(row?.status) >= 200 && Number(row?.status) < 300)) return false; + } else if (typeof filter.status === "number" || (typeof filter.status === "string" && !isNaN(Number(filter.status)))) { + if (Number(row?.status) !== Number(filter.status)) return false; + } + + if (filter.model && !matchesSearch(row?.model || "", String(filter.model))) { + return false; + } + if (filter.provider && !matchesSearch(row?.provider || "", String(filter.provider))) { + return false; + } + if (filter.account && !matchesSearch(row?.account || "", String(filter.account))) { + return false; + } + if (filter.apiKey && !matchesSearch(row?.apiKeyName || "", String(filter.apiKey))) { + return false; + } + if (filter.combo && !matchesSearch(row?.comboName || "", String(filter.combo))) { + return false; + } + if (filter.correlationId && !matchesSearch(row?.correlationId || "", String(filter.correlationId))) { + return false; + } + if (filter.search) { + const term = String(filter.search); + const haystack = [ + row?.model, + row?.provider, + row?.providerDisplay, + row?.account, + row?.apiKeyName, + row?.comboName, + row?.correlationId, + row?.error, + row?.path, + ] + .filter(Boolean) + .join(" "); + if (!matchesSearch(haystack, term)) return false; + } + + return true; +} + export function buildCallLogListRows({ logs, connections, @@ -174,15 +234,8 @@ export async function GET(request: Request) { completedDetails: getCompletedDetails().values(), }); - // When correlationId filter is set, also filter in-memory entries - // (active + completed) that don't match — getCallLogs already filters - // the DB rows but activeEntries/completedEntries bypass it. - if (filter.correlationId) { - const cid = filter.correlationId; - return NextResponse.json(rows.filter((r: any) => matchesSearch(r.correlationId || "", cid))); - } - - return NextResponse.json(rows); + const filtered = rows.filter((r: any) => rowMatchesFilter(r, filter)); + return NextResponse.json(filtered); } catch (error) { console.error("[API ERROR] /api/usage/call-logs failed:", error); return NextResponse.json({ error: "Failed to fetch call logs" }, { status: 500 }); diff --git a/src/app/api/v1/rerank/route.ts b/src/app/api/v1/rerank/route.ts index bf9da386eb4a..04fb1632dc6c 100644 --- a/src/app/api/v1/rerank/route.ts +++ b/src/app/api/v1/rerank/route.ts @@ -10,11 +10,15 @@ import { HTTP_STATUS } from "@omniroute/open-sse/config/constants.ts"; import { enforceApiKeyPolicy } from "@/shared/utils/apiKeyPolicy"; import { v1RerankSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; -import { getCachedProviderNodes } from "@/lib/localDb"; +import { getCachedProviderNodes } from "@/lib/db/readCache"; import { isAllRateLimitedCredentials, rateLimitedProviderResponse, } from "@/app/api/v1/_shared/rateLimit"; +import { saveCallLog } from "@/lib/usageDb"; +import { attachOmniRouteMetaHeaders } from "@/domain/omnirouteResponseMeta"; +import { generateRequestId } from "@/shared/utils/requestId"; +import { CORS_HEADERS } from "@omniroute/open-sse/utils/cors.ts"; /** * Handle CORS preflight @@ -121,6 +125,8 @@ async function postHandler(request, context) { return_documents: body.return_documents, credentials, connectionId: (credentials as { connectionId?: string } | null)?.connectionId || null, + apiKeyId: policy.apiKeyInfo?.id || null, + apiKeyName: policy.apiKeyInfo?.name || null, }); if (response?.ok) { await clearRecoveredProviderState(credentials); @@ -148,8 +154,9 @@ async function postHandler(request, context) { } const token = credentials?.apiKey || credentials?.accessToken; + const startTime = Date.now(); try { - const res = await fetch(localProvider.baseUrl, { + let res = await fetch(localProvider.baseUrl, { method: "POST", headers: { "Content-Type": "application/json", @@ -164,19 +171,110 @@ async function postHandler(request, context) { }), }); + // Some local providers (e.g. Infinity, TEI) mount at /rerank rather than /v1/rerank + if (res.status === 404 && localProvider.baseUrl.endsWith("/v1/rerank")) { + const fallbackUrl = localProvider.baseUrl.replace(/\/v1\/rerank$/, "/rerank"); + try { + const fallbackRes = await fetch(fallbackUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }, + body: JSON.stringify({ + model: localModel, + query: body.query, + documents: body.documents, + top_n: body.top_n || body.documents.length, + return_documents: body.return_documents !== false, + }), + }); + if (fallbackRes.ok || fallbackRes.status !== 404) { + res = fallbackRes; + } + } catch { + // retain original 404 response if fallback fetch fails + } + } + if (!res.ok) { const errData = await res.json().catch(() => ({})); - return errorResponse( - res.status, - errData.message || errData.detail || `Provider returned HTTP ${res.status}` - ); + const errorMessage = + errData.message || errData.detail || `Provider returned HTTP ${res.status}`; + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: res.status, + model: body.model, + provider: prefix, + connectionId: + (credentials as { connectionId?: string } | null)?.connectionId || undefined, + duration: Date.now() - startTime, + requestBody: { + model: body.model, + query: body.query, + documents: body.documents, + top_n: body.top_n, + return_documents: body.return_documents, + }, + responseBody: errData, + error: errorMessage, + apiKeyId: policy.apiKeyInfo?.id || undefined, + apiKeyName: policy.apiKeyInfo?.name || undefined, + }).catch(() => {}); + return errorResponse(res.status, errorMessage); } const data = await res.json(); - return Response.json(data, { - headers: {}, + const latencyMs = Date.now() - startTime; + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: 200, + model: body.model, + provider: prefix, + connectionId: + (credentials as { connectionId?: string } | null)?.connectionId || undefined, + duration: latencyMs, + tokens: { prompt_tokens: 0, completion_tokens: 0 }, + requestBody: { + model: body.model, + query: body.query, + documents: body.documents, + top_n: body.top_n, + return_documents: body.return_documents, + }, + responseBody: data, + apiKeyId: policy.apiKeyInfo?.id || undefined, + apiKeyName: policy.apiKeyInfo?.name || undefined, + }).catch(() => {}); + + const headers = new Headers({ ...CORS_HEADERS, "Content-Type": "application/json" }); + attachOmniRouteMetaHeaders(headers, { + provider: prefix, + model: localModel, + costUsd: 0, + latencyMs, + requestId: generateRequestId(), + }); + return new Response(JSON.stringify(data), { + status: 200, + headers, }); } catch (err: any) { + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: 500, + model: body.model, + provider: prefix, + connectionId: + (credentials as { connectionId?: string } | null)?.connectionId || undefined, + duration: Date.now() - startTime, + error: err.message, + apiKeyId: policy.apiKeyInfo?.id || undefined, + apiKeyName: policy.apiKeyInfo?.name || undefined, + }).catch(() => {}); return errorResponse(500, `Rerank request failed: ${err.message}`); } } diff --git a/src/app/api/v1/search/route.ts b/src/app/api/v1/search/route.ts index adb888dff778..313b72dece0c 100644 --- a/src/app/api/v1/search/route.ts +++ b/src/app/api/v1/search/route.ts @@ -58,9 +58,7 @@ export async function OPTIONS() { export async function GET() { const settings = await getSettings().catch(() => ({} as any)); const blockedProviders = settings?.blockedProviders || []; - const providers = getAllSearchProviders().filter( - (p) => !isProviderBlockedByIdOrAlias(p.id, blockedProviders) - ); + const providers = getAllSearchProviders(blockedProviders); const timestamp = Math.floor(Date.now() / 1000); const data = providers.map((p) => ({ diff --git a/src/domain/configAudit.ts b/src/domain/configAudit.ts index c2701b2a21f7..873d3b046a5f 100644 --- a/src/domain/configAudit.ts +++ b/src/domain/configAudit.ts @@ -13,6 +13,8 @@ * - Optional human notes */ +import { getDbInstance } from "../lib/db/core"; + /** Types of configuration entities that can be audited */ export type AuditTarget = "provider" | "combo" | "policy" | "connection" | "settings"; @@ -72,10 +74,8 @@ export interface ConfigSnapshot { data: Record; } -// ── In-memory store ────────────────────────────────────────────────────────── -// In production, persist to SQLite alongside other domain state. +// ── SQLite-backed store ─────────────────────────────────────────────────────── -let auditLog: ConfigAuditEntry[] = []; let idCounter = 0; function generateId(): string { @@ -85,6 +85,40 @@ function generateId(): string { return `audit-${ts}-${seq}`; } +function db() { + return getDbInstance(); +} + +interface ConfigAuditRow { + id: string; + timestamp: string; + action: string; + target: string; + target_id: string; + target_name: string; + before_json: string | null; + after_json: string | null; + diff_json: string; + source: string; + note: string | null; +} + +function rowToEntry(row: ConfigAuditRow): ConfigAuditEntry { + return { + id: row.id, + timestamp: row.timestamp, + action: row.action as AuditAction, + target: row.target as AuditTarget, + targetId: row.target_id, + targetName: row.target_name, + before: row.before_json === null ? null : (JSON.parse(row.before_json) as Record | null), + after: row.after_json === null ? null : (JSON.parse(row.after_json) as Record | null), + source: row.source as AuditSource, + diff: JSON.parse(row.diff_json) as ConfigDiff, + note: row.note, + }; +} + /** * Compute a structured diff between two configuration states. */ @@ -159,12 +193,24 @@ export function recordChange( note: note ?? null, }; - auditLog.push(entry); - - // Keep log bounded (max 1000 entries in memory) - if (auditLog.length > 1000) { - auditLog = auditLog.slice(-1000); - } + db().prepare( + `INSERT INTO config_audit_log + (id, timestamp, action, target, target_id, target_name, before_json, after_json, diff_json, source, note) + VALUES + (@id, @timestamp, @action, @target, @targetId, @targetName, @beforeJson, @afterJson, @diffJson, @source, @note)` + ).run({ + id: entry.id, + timestamp: entry.timestamp, + action: entry.action, + target: entry.target, + targetId: entry.targetId, + targetName: entry.targetName, + beforeJson: before === null ? null : JSON.stringify(before), + afterJson: after === null ? null : JSON.stringify(after), + diffJson: JSON.stringify(entry.diff), + source: entry.source, + note: entry.note, + }); return entry; } @@ -181,42 +227,57 @@ export function getAuditLog(options?: { limit?: number; offset?: number; }): { entries: ConfigAuditEntry[]; total: number } { - let filtered = auditLog; + const where: string[] = []; + const params: Record = {}; if (options?.target) { - filtered = filtered.filter((e) => e.target === options.target); + where.push("target = @target"); + params.target = options.target; } if (options?.targetId) { - filtered = filtered.filter((e) => e.targetId === options.targetId); + where.push("target_id = @targetId"); + params.targetId = options.targetId; } if (options?.action) { - filtered = filtered.filter((e) => e.action === options.action); + where.push("action = @action"); + params.action = options.action; } if (options?.source) { - filtered = filtered.filter((e) => e.source === options.source); + where.push("source = @source"); + params.source = options.source; } if (options?.since) { - filtered = filtered.filter((e) => e.timestamp >= options.since!); + where.push("timestamp >= @since"); + params.since = options.since; } - const total = filtered.length; + const whereSql = where.length > 0 ? `WHERE ${where.join(" AND ")}` : ""; - // Sort newest first - filtered = [...filtered].sort((a, b) => b.timestamp.localeCompare(a.timestamp)); + const totalRow = db() + .prepare(`SELECT COUNT(*) AS c FROM config_audit_log ${whereSql}`) + .get(params) as { c: number }; + const total = totalRow.c; - // Paginate const offset = options?.offset ?? 0; const limit = options?.limit ?? 50; - filtered = filtered.slice(offset, offset + limit); - return { entries: filtered, total }; + const rows = db() + .prepare( + `SELECT * FROM config_audit_log ${whereSql} ORDER BY datetime(timestamp) DESC, id DESC LIMIT @limit OFFSET @offset` + ) + .all({ ...params, limit, offset }) as ConfigAuditRow[]; + + return { entries: rows.map(rowToEntry), total }; } /** * Get a specific audit entry by ID. */ export function getAuditEntry(id: string): ConfigAuditEntry | null { - return auditLog.find((e) => e.id === id) ?? null; + const row = db() + .prepare("SELECT * FROM config_audit_log WHERE id = @id") + .get({ id }) as ConfigAuditRow | undefined; + return row ? rowToEntry(row) : null; } /** @@ -260,19 +321,23 @@ export function getAuditSummary(): { const byAction: Record = {}; const bySource: Record = {}; - for (const entry of auditLog) { - byTarget[entry.target] = (byTarget[entry.target] || 0) + 1; - byAction[entry.action] = (byAction[entry.action] || 0) + 1; - bySource[entry.source] = (bySource[entry.source] || 0) + 1; + const rows = db() + .prepare("SELECT * FROM config_audit_log ORDER BY datetime(timestamp) DESC, id DESC") + .all() as ConfigAuditRow[]; + + for (const row of rows) { + byTarget[row.target] = (byTarget[row.target] || 0) + 1; + byAction[row.action] = (byAction[row.action] || 0) + 1; + bySource[row.source] = (bySource[row.source] || 0) + 1; } return { - totalEntries: auditLog.length, + totalEntries: rows.length, byTarget, byAction, bySource, - oldestEntry: auditLog.length > 0 ? auditLog[0].timestamp : null, - newestEntry: auditLog.length > 0 ? auditLog[auditLog.length - 1].timestamp : null, + oldestEntry: rows.length > 0 ? rows[rows.length - 1].timestamp : null, + newestEntry: rows.length > 0 ? rows[0].timestamp : null, }; } @@ -280,6 +345,6 @@ export function getAuditSummary(): { * Reset the audit log. Useful for testing. */ export function resetAuditLog(): void { - auditLog = []; + db().prepare("DELETE FROM config_audit_log").run(); idCounter = 0; } diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index ec091011e9d9..930f5b595875 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -5692,8 +5692,8 @@ "aggregatorsGateways": "Aggregators Gateways", "enterpriseCloud": "Enterprise & Cloud", "apiFormatLabel": "Api Format Label", - "apiKeyOptionalHint": "Api Key Optional Hint", - "apiKeyOptionalLabel": "Api Key Optional Label", + "apiKeyOptionalHint": "Leave empty if your local setup or provider does not require authentication.", + "apiKeyOptionalLabel": "API Key (optional)", "apiRegionChina": "Api Region China", "apiRegionHint": "Api Region Hint", "apiRegionInternational": "Api Region International", @@ -13869,5 +13869,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "Cheaper Inference is an OmniRoute Open Source Friend", + "description": "A cost-ranked gateway reselling dozens of frontier models behind one OpenAI-compatible endpoint — routing each request to the cheapest eligible provider, never above list price.", + "cta": "Get an API Key", + "partnerLinkNote": "Partner link", + "dismissAriaLabel": "Dismiss" } } diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 11a3606a54cd..1921f3fba7fb 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -13845,5 +13845,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "A Cheaper Inference é uma Amiga do Código Aberto do OmniRoute", + "description": "Um gateway com custo ordenado que revende dezenas de modelos de fronteira num único endpoint compatível com OpenAI — roteando cada requisição ao provedor elegível mais barato, nunca acima do preço de tabela.", + "cta": "Obter uma Chave de API", + "partnerLinkNote": "Link de parceiro", + "dismissAriaLabel": "Dispensar" } } diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index a17c1fc52c41..e04b20666139 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -6269,6 +6269,20 @@ "webSessionGuideStep3": "Sao chép thông tin xác thực được yêu cầu từ tên miền riêng của nhà cung cấp. Đối với cookie, chỉ sao chép giá trị tiêu đề Cookie và bỏ qua Cookie:.", "webSessionGuideStep3Manual": "Cách thủ công: mở công cụ dành cho nhà phát triển của trình duyệt (F12 → Network), tải lại trang, mở một yêu cầu đã xác thực và sao chép giá trị tiêu đề Cookie trong Request Headers — bỏ tiền tố Cookie:.", "webSessionGuideStep4": "Dán vào đây và kiểm tra kết nối. Nếu nó ngừng hoạt động, hãy đăng nhập lại và thay thế bằng một giá trị mới.", + "harImportButtonLabel": "Nhập tệp .har", + "harImportButtonBusy": "Đang nhập…", + "harImportButtonHint": "Xuất từ thẻ Network của DevTools sau khi gửi ít nhất một tin nhắn chat.", + "harImportStatusValid": "Đã nhập — hợp lệ trong ~{minutes} phút.", + "harImportStatusExpiringSoon": "Đã nhập — chỉ còn hợp lệ ~{minutes} phút nữa.", + "harImportStatusExpired": "Đã nhập, nhưng token này đã hết hạn ({minutes} phút trước) — hãy xuất một HAR mới.", + "harImportStatusUnknownExpiry": "Đã nhập. Không đọc được thời hạn.", + "harImportErrorNotJson": "Tệp đó không phải JSON hợp lệ — có đúng là bản xuất .har không?", + "harImportErrorNoEntries": "HAR này không có mục network nào được ghi lại.", + "harImportErrorNoChathubUrl": "Không tìm thấy kết nối Copilot chat trong HAR này. Hãy gửi ít nhất một tin nhắn chat trong m365.cloud.microsoft trước khi xuất.", + "harImportErrorUnparsableUrl": "Tìm thấy kết nối chat, nhưng không đọc được URL của nó.", + "harImportErrorMissingFields": "Tìm thấy kết nối chat, nhưng token bị thiếu trong đó.", + "harImportErrorReadFailed": "Không đọc được tệp đó.", + "harImportErrorUnknown": "Không trích xuất được thông tin xác thực từ tệp HAR đó.", "webSessionSecurityHint": "Hãy coi đây như mật khẩu: nó có thể truy cập vào tài khoản web đã đăng nhập của bạn cho đến khi hết hạn hoặc bị thu hồi.", "webNoAuthGuideTitle": "Không yêu cầu thông tin xác thực", "webNoAuthGuideBody": "{provider} không cần khóa API hoặc cookie. Lưu kết nối để sử dụng endpoint web miễn phí của nó.", @@ -12207,7 +12221,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." + "description": "Đăng ký, liệt kê, kiểm thử và xoá các endpoint webhook. Cấu hình đăng ký sự kiện (request.completed, request.failed, quota.exceeded, v.v.) và quản lý thử lại giao hàng." }, "omni-mcp": { "name": "Máy chủ MCP", diff --git a/src/lib/cli-helper/config-generator/opencode.ts b/src/lib/cli-helper/config-generator/opencode.ts index a2329968abc3..e65206dc5574 100644 --- a/src/lib/cli-helper/config-generator/opencode.ts +++ b/src/lib/cli-helper/config-generator/opencode.ts @@ -280,10 +280,9 @@ function buildModelEntry( } // Resolve the context window. Honor an explicit user override, then fall - // back to the catalog. We do NOT synthesize a default — if the catalog - // is unaware of a model's window, the opencode.json will simply omit - // `limit.context` for that model and OpenCode's own heuristics apply. - // (OpenCode v1 defaults to 128K when `limit.context` is missing.) + // back to the catalog. If the catalog is unaware of a model's window, we + // fall back to a safe default (128K) so OpenCode's v1 provider schema + // validator never rejects the config with a missing key error (#11035). const userLimit = existing?.limit?.context; const catalogLimit = catalog ? resolveContextLength(catalog) : undefined; const context = typeof userLimit === "number" && userLimit > 0 ? userLimit : catalogLimit; @@ -292,9 +291,6 @@ function buildModelEntry( // Use the catalog's max_output_tokens when available; otherwise fall // back to the user's existing `limit.output` and finally to a small // default (8K) so OpenCode never errors on a totally missing output cap. - // We do NOT default context — context is a property of the model and - // we have no business guessing. Output is a per-request setting and a - // small default is harmless when truly unknown. const userOutput = existing?.limit?.output; const catalogOutput = catalog && typeof catalog.max_output_tokens === "number" && catalog.max_output_tokens > 0 @@ -411,7 +407,7 @@ export interface GenerateOpencodeOptions { /** * If `true` (default), the generator fetches the live `/v1/models` catalog * so every model entry has an explicit `limit.context`. The catalog is the - * single source of truth for context windows; we never invent defaults. + * primary source of truth for context windows, falling back to 128K when unknown. * * When the catalog request fails, the generator throws — opencode.json must * not be emitted with stale or fabricated values. The CLI can catch the @@ -426,7 +422,7 @@ export interface GenerateOpencodeOptions { /** * Generate a full `opencode.json` document for OmniRoute. The catalog is the - * single source of truth for context windows — we never hardcode values. + * primary source of truth for context windows, with a 128K fallback when unknown. * * Behavior: * - Preserves the user's existing provider name, npm, options, and @@ -436,8 +432,8 @@ export interface GenerateOpencodeOptions { * - For each catalog model id the user did NOT have, a new entry is * added with `limit.context` populated when the catalog has it. * - If the catalog has no context for a model AND the user has no - * override, the model is emitted WITHOUT a `limit.context` field. - * OpenCode's own heuristic (typically 128K) applies. + * override, a safe default (128K) is emitted so OpenCode's schema validator + * does not reject the model. * - Throws if the catalog fetch fails — the user must fix the upstream * before we can generate a reliable opencode.json. */ diff --git a/src/lib/config/runtimeSettings.ts b/src/lib/config/runtimeSettings.ts index dc5e9d720dc8..bf615749bae6 100644 --- a/src/lib/config/runtimeSettings.ts +++ b/src/lib/config/runtimeSettings.ts @@ -1,5 +1,9 @@ import { clearHealthCheckLogCache } from "@/lib/tokenHealthCheck"; import { setCustomBannedSignals } from "@omniroute/open-sse/services/accountFallback.ts"; +import { + setOperatorProviderErrorRules, + type OperatorProviderErrorRule, +} from "@omniroute/open-sse/config/providerErrorRules.ts"; import { isAutomatedTestProcess } from "@/shared/utils/testProcess"; type JsonRecord = Record; @@ -46,6 +50,7 @@ interface RuntimeSettingsSnapshot { systemTransforms: unknown; authzBypass: AuthzBypassSnapshot; customBannedSignals: string[]; + providerErrorRules: Record | null; } // Default bypass policy: kill-switch on, `/api/mcp/` bypassable. Mirrors the @@ -72,6 +77,7 @@ const DEFAULT_RUNTIME_SETTINGS_SNAPSHOT: RuntimeSettingsSnapshot = { systemTransforms: null, authzBypass: DEFAULT_AUTHZ_BYPASS_SNAPSHOT, customBannedSignals: [], + providerErrorRules: null, }; let lastAppliedSnapshot: RuntimeSettingsSnapshot | null = null; @@ -138,6 +144,34 @@ function normalizeStringArray(value: unknown): string[] { ); } +/** + * Defensive shape-check of operator-declared error rules pulled from settings. + * The settings schema already validates this on write; this guard prevents a + * malformed stored value (or an unexpected shape) from crashing the + * error-classification hot path. Returns null when the value is missing or not + * a record of non-empty rule arrays. + */ +function normalizeOperatorProviderErrorRules( + value: unknown +): Record | null { + if (value === null || typeof value !== "object") return null; + const record = value as Record; + const result: Record = {}; + for (const [provider, list] of Object.entries(record)) { + if (!Array.isArray(list) || list.length === 0) continue; + const rules = list.filter( + (entry): entry is OperatorProviderErrorRule => + !!entry && + typeof entry === "object" && + typeof (entry as OperatorProviderErrorRule).status === "number" && + typeof (entry as OperatorProviderErrorRule).match === "string" && + typeof (entry as OperatorProviderErrorRule).scope === "string" + ); + if (rules.length > 0) result[provider.toLowerCase()] = rules; + } + return Object.keys(result).length > 0 ? result : null; +} + function normalizeStringRecord(value: unknown): Record { const record = toRecord(parseStoredJson(value, "modelAliases")); const entries = Object.entries(record) @@ -244,6 +278,7 @@ export function buildRuntimeSettingsSnapshot( systemTransforms: parseStoredJson(settings.systemTransforms, "systemTransforms"), authzBypass: normalizeAuthzBypass(settings), customBannedSignals: normalizeStringArray(settings.customBannedSignals), + providerErrorRules: normalizeOperatorProviderErrorRules(settings.providerErrorRules), }; } @@ -540,6 +575,13 @@ export async function applyRuntimeSettings( markChanged("bannedSignals"); } + if ( + force || + hasChanged(currentSnapshot.providerErrorRules, previousSnapshot.providerErrorRules) + ) { + setOperatorProviderErrorRules(currentSnapshot.providerErrorRules ?? undefined); + } + lastAppliedSnapshot = currentSnapshot; return changes; } diff --git a/src/lib/db/adapters/driverFactory.ts b/src/lib/db/adapters/driverFactory.ts index 64c06153dbe6..304b334bd8eb 100644 --- a/src/lib/db/adapters/driverFactory.ts +++ b/src/lib/db/adapters/driverFactory.ts @@ -224,7 +224,10 @@ export function createSyncDriverFactory(load: DriverLoader, betterSqliteProbe?: const bunOptions: Record = {}; if (options?.readonly === true) bunOptions.readonly = true; if (options?.create === false && filePath !== ":memory:") bunOptions.create = false; - const db = new Database(filePath, bunOptions); + const db = + Object.keys(bunOptions).length > 0 + ? new Database(filePath, bunOptions) + : new Database(filePath); return createBunSqliteAdapter(db, filePath); } catch (err) { logSwallowedDriverError("bun:sqlite", err); diff --git a/src/lib/db/cleanup.ts b/src/lib/db/cleanup.ts index 617bf4c22642..60e288bf4b47 100644 --- a/src/lib/db/cleanup.ts +++ b/src/lib/db/cleanup.ts @@ -193,6 +193,31 @@ export async function cleanupMcpAudit(): Promise { return result; } +/** + * Clean up old config_audit_log based on retention settings. + */ +export async function cleanupConfigAudit(retentionDays = getRetentionSettings().configAudit): Promise { + const db = getDbInstance(); + const result: CleanupResult = { deleted: 0, errors: 0 }; + + try { + const stmt = db.prepare( + "DELETE FROM config_audit_log WHERE datetime(timestamp) < datetime('now', '-' || ? || ' days')" + ); + const runResult = stmt.run(String(retentionDays)); + result.deleted = runResult.changes; + + console.log( + `[Cleanup] Deleted ${result.deleted} config_audit_log older than ${retentionDays} days` + ); + } catch (err: unknown) { + console.error("[Cleanup] Error cleaning config_audit_log:", err); + result.errors++; + } + + return result; +} + /** * Clean up old a2a_task_events based on retention settings. */ @@ -420,6 +445,7 @@ export async function runAutoCleanup(): Promise<{ usageHistory: await cleanupUsageHistory(), compressionAnalytics: await cleanupCompressionAnalytics(), mcpAudit: await cleanupMcpAudit(), + configAudit: await cleanupConfigAudit(), a2aEvents: await cleanupA2aEvents(), memoryEntries: await cleanupMemoryEntries(), domainCostHistory: await cleanupDomainCostHistory(), diff --git a/src/lib/db/databaseSettings.ts b/src/lib/db/databaseSettings.ts index e18f2a66bd9a..0a12729242cd 100644 --- a/src/lib/db/databaseSettings.ts +++ b/src/lib/db/databaseSettings.ts @@ -46,6 +46,7 @@ const LEGACY_FLAT_KEYS: { quotaSnapshots: ["quotaSnapshots"], compressionAnalytics: ["compressionAnalytics"], mcpAudit: ["mcpAudit"], + configAudit: ["configAudit"], a2aEvents: ["a2aEvents"], callLogs: ["callLogs"], usageHistory: ["usageHistory"], diff --git a/src/lib/db/migrations/046_database_settings.sql b/src/lib/db/migrations/046_database_settings.sql index 57fd15c903be..6fb9864390a5 100644 --- a/src/lib/db/migrations/046_database_settings.sql +++ b/src/lib/db/migrations/046_database_settings.sql @@ -25,6 +25,7 @@ INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSetting INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'quotaSnapshots', '90'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'compressionAnalytics', '30'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'mcpAudit', '30'); +INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'configAudit', '30'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'a2aEvents', '30'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'callLogs', '90'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'usageHistory', '365'); diff --git a/src/lib/db/migrations/161_config_audit_log.sql b/src/lib/db/migrations/161_config_audit_log.sql new file mode 100644 index 000000000000..0aad91ace28e --- /dev/null +++ b/src/lib/db/migrations/161_config_audit_log.sql @@ -0,0 +1,15 @@ +CREATE TABLE IF NOT EXISTS config_audit_log ( + id TEXT PRIMARY KEY, + timestamp TEXT NOT NULL, + action TEXT NOT NULL, + target TEXT NOT NULL, + target_id TEXT NOT NULL, + target_name TEXT NOT NULL, + before_json TEXT, + after_json TEXT, + diff_json TEXT NOT NULL, + source TEXT NOT NULL, + note TEXT +); +CREATE INDEX IF NOT EXISTS idx_config_audit_log_target_created ON config_audit_log(target, timestamp); +CREATE INDEX IF NOT EXISTS idx_config_audit_log_created ON config_audit_log(timestamp); diff --git a/src/lib/db/migrations/162_remove_hackclub_provider.sql b/src/lib/db/migrations/162_remove_hackclub_provider.sql new file mode 100644 index 000000000000..4dd20904e96e --- /dev/null +++ b/src/lib/db/migrations/162_remove_hackclub_provider.sql @@ -0,0 +1,21 @@ +-- 162_remove_hackclub_provider.sql +-- Hack Club AI provider was removed from OmniRoute at the request of Hack Club's +-- maintainers (#11118). Clean up any locally stored configuration for it. +-- Historical request and usage records are intentionally preserved under the +-- provider identity that existed when they were written. + +DELETE FROM provider_connections +WHERE provider = 'hackclub'; + +DELETE FROM registered_keys +WHERE provider = 'hackclub'; + +DELETE FROM provider_key_limits +WHERE provider = 'hackclub'; + +DELETE FROM discovery_results +WHERE provider_id = 'hackclub'; + +DELETE FROM key_value +WHERE namespace = 'customModels' + AND key = 'hackclub'; diff --git a/src/lib/db/providers.ts b/src/lib/db/providers.ts index 3b02933e1135..fbab6817fd69 100644 --- a/src/lib/db/providers.ts +++ b/src/lib/db/providers.ts @@ -627,15 +627,26 @@ export async function createProviderConnection(data: JsonRecord) { // to no-overrides) keeps the field present on the returned object so the // UI can tell "field was read, no overrides" apart from "field absent." if ("quotaWindowThresholds" in connection) { - connection.quotaWindowThresholds = sanitizeQuotaWindowThresholds( - connection.quotaWindowThresholds - ); + const result = sanitizeQuotaWindowThresholds(connection.quotaWindowThresholds); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist quotaWindowThresholds with rejected keys: ${result.rejected.join(", ")}` + ); + } + connection.quotaWindowThresholds = result.sanitized; } // Same sanitization for rateLimitOverrides — keep in-memory representation - // in sync with what gets persisted. + // in sync with what gets persisted. Reject (don't silently drop) invalid + // keys/values so a direct DB writer can't lose operator intent. if ("rateLimitOverrides" in connection) { - connection.rateLimitOverrides = sanitizeRateLimitOverrides(connection.rateLimitOverrides); + const result = sanitizeRateLimitOverrides(connection.rateLimitOverrides); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist rateLimitOverrides with rejected keys: ${result.rejected.join(", ")}` + ); + } + connection.rateLimitOverrides = result.sanitized; } _insertConnectionRow(db, encryptConnectionFields({ ...connection })); @@ -849,13 +860,24 @@ export async function updateProviderConnection(id: string, data: JsonRecord) { // Mirror the sanitization the create path applies — keep the returned // object in lockstep with what we persist. if ("quotaWindowThresholds" in merged) { - const sanitized = sanitizeQuotaWindowThresholds(merged.quotaWindowThresholds); + const result = sanitizeQuotaWindowThresholds(merged.quotaWindowThresholds); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist quotaWindowThresholds with rejected keys: ${result.rejected.join(", ")}` + ); + } // For updates we always carry the key forward (even as null) so the read - // path surfaces the cleared state to callers that just patched it. - merged.quotaWindowThresholds = sanitized; + // path surfaces the cleared state to callers that merged it. + merged.quotaWindowThresholds = result.sanitized; } if ("rateLimitOverrides" in merged) { - merged.rateLimitOverrides = sanitizeRateLimitOverrides(merged.rateLimitOverrides); + const result = sanitizeRateLimitOverrides(merged.rateLimitOverrides); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist rateLimitOverrides with rejected keys: ${result.rejected.join(", ")}` + ); + } + merged.rateLimitOverrides = result.sanitized; } const existingRecord = toRecord(existing); diff --git a/src/lib/db/providers/columns.ts b/src/lib/db/providers/columns.ts index b32f653f7c6c..f06f948cd212 100644 --- a/src/lib/db/providers/columns.ts +++ b/src/lib/db/providers/columns.ts @@ -64,20 +64,37 @@ export function normalizeBooleanColumn(value: unknown, fallback: boolean): boole return fallback; } +// Result of sanitizing a per-connection overrides/threshold map. `sanitized` +// is the cleaned value (or null when it collapses to nothing); `rejected` +// lists every key that was refused so callers can fail loudly +// instead of silently dropping the operator's input. +export type SanitizeResult = { + sanitized: Record | null; + rejected: string[]; +}; + // Sanitize the per-connection rate limit overrides map: keep only known -// fields with valid numeric values. Called once at each write-path boundary. -export function sanitizeRateLimitOverrides(value: unknown): Record | null { - if (value === null || value === undefined) return null; - if (typeof value !== "object" || Array.isArray(value)) return null; +// fields with valid non-negative integer values. Called once at each +// write-path boundary. Unknown keys and invalid values go into `rejected` +// rather than being dropped in silence. +export function sanitizeRateLimitOverrides(value: unknown): SanitizeResult { + if (value === null || value === undefined) return { sanitized: null, rejected: [] }; + if (typeof value !== "object" || Array.isArray(value)) return { sanitized: null, rejected: [] }; const allowedKeys = new Set(["rpm", "tpm", "tpd", "minTime", "maxConcurrent"]); + const rejected: string[] = []; const map: Record = {}; for (const [key, v] of Object.entries(value as Record)) { - if (!allowedKeys.has(key)) continue; + if (!allowedKeys.has(key)) { + rejected.push(key); + continue; + } if (typeof v === "number" && Number.isInteger(v) && v >= 0) { map[key] = v; + } else { + rejected.push(key); } } - return Object.keys(map).length === 0 ? null : map; + return { sanitized: Object.keys(map).length === 0 ? null : map, rejected }; } // Serialize an already-sanitized map for SQLite TEXT storage. @@ -91,20 +108,29 @@ export function toRecord(value: unknown): JsonRecord { return value && typeof value === "object" ? (value as JsonRecord) : {}; } -// Sanitize the per-window threshold map: keep only 0-100 integer values. -// Called once at each write-path boundary (createProviderConnection + -// updateProviderConnection) so both the in-memory return and the persisted -// row share the same shape. Serialization below trusts this output. -export function sanitizeQuotaWindowThresholds(value: unknown): Record | null { - if (value === null || value === undefined) return null; - if (typeof value !== "object" || Array.isArray(value)) return null; +// Sanitize the per-window threshold map: keep only 0-100 integer values with +// keys no longer than 64 chars. Called once at each write-path boundary +// (createProviderConnection + updateProviderConnection) so both the in-memory +// return and the persisted row share the same shape. Serialization below +// trusts this output. Invalid keys/values go into `rejected` rather than being +// dropped in silence. +export function sanitizeQuotaWindowThresholds(value: unknown): SanitizeResult { + if (value === null || value === undefined) return { sanitized: null, rejected: [] }; + if (typeof value !== "object" || Array.isArray(value)) return { sanitized: null, rejected: [] }; + const rejected: string[] = []; const map: Record = {}; for (const [key, v] of Object.entries(value as Record)) { + if (key.length > 64) { + rejected.push(key); + continue; + } if (typeof v === "number" && Number.isInteger(v) && v >= 0 && v <= 100) { map[key] = v; + } else { + rejected.push(key); } } - return Object.keys(map).length === 0 ? null : map; + return { sanitized: Object.keys(map).length === 0 ? null : map, rejected }; } export function toStringOrNull(value: unknown): string | null { diff --git a/src/lib/memory/injection.ts b/src/lib/memory/injection.ts index 9fdd2bbbf108..d4d8ead7f7c6 100644 --- a/src/lib/memory/injection.ts +++ b/src/lib/memory/injection.ts @@ -65,7 +65,11 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * * Populated with the Xiaomi MiMo endpoint (provider id `xiaomi-mimo`, registry * alias `mimo`, serving mimo-v2.5) confirmed live to 400 on a non-first system - * message. Add other providers here only when they are documented as strict. + * message, and the TokenRouter gateway (provider id `tokenrouter`), confirmed + * live on 2026-08-22 to reject mid-array system messages — including the + * compression notice spliced by purifyHistory() before that splice was fixed to + * merge into the leading system message. Add other providers here only when + * they are documented as strict. * * Self-hosted deployments can extend this list without a source change via * OMNIROUTE_STRICT_SYSTEM_PROVIDERS (comma-separated provider ids, @@ -73,7 +77,7 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * self-hosted Qwen3.5+/3.6 model, whose chat template enforces the same * single-leading-system-message constraint as xiaomi-mimo. */ -const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo"]); +const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo", "tokenrouter"]); /** * Parses OMNIROUTE_STRICT_SYSTEM_PROVIDERS into a normalized id list. diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 3828047b90b9..b7aa2aa21643 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -130,6 +130,11 @@ function uniqueStrings(values: Array) { ]; } +export function isGlmFamilyModel(modelId: string, displayName = ""): boolean { + const glmFamilyPattern = /(?:^|[/@:_. -])glm(?=$|[-._ /@:](?:z)?\d|\d)/i; + return glmFamilyPattern.test(modelId) || glmFamilyPattern.test(displayName); +} + function toQualifiedId( providerAlias: string | null, provider: string | null, @@ -477,11 +482,16 @@ export function enrichCatalogModelEntry( ? declaredEffortTiers : sourceDeclaresThinking ? undefined - : extendCodexGpt56EffortValues( - metadata.provider, - metadata.model, - CANONICAL_EFFORT_VALUES - ); + : // #10963: GLM-family models never inherit generic OpenAI tiers — an + // explicit empty list is authoritative unless a provider-declared + // contract exists (handled by declaredEffortTiers above). + isGlmFamilyModel(metadata.model, metadata.displayName) + ? [] + : extendCodexGpt56EffortValues( + metadata.provider, + metadata.model, + CANONICAL_EFFORT_VALUES + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -502,7 +512,9 @@ export function enrichCatalogModelEntry( // #6241: surface thinking support + the canonical effort tiers so the frontend can // render the effort/thinking toggles. `thinking` is kept for back-compat; `supportsThinking` // is the explicit flag and `effort_tiers` lists the selectable reasoning levels - // (only when the model actually supports thinking). + // (only when the model actually supports thinking). An explicit empty registry list + // is authoritative; GLM models also require a provider-declared contract instead of + // inheriting generic OpenAI effort tiers. ...(typeof metadata.capabilities.supportsThinking === "boolean" ? { thinking: metadata.capabilities.supportsThinking, diff --git a/src/lib/monitoring/comboHealthAutopilot.ts b/src/lib/monitoring/comboHealthAutopilot.ts index 71a104948375..24eba32a634c 100644 --- a/src/lib/monitoring/comboHealthAutopilot.ts +++ b/src/lib/monitoring/comboHealthAutopilot.ts @@ -16,6 +16,7 @@ import type { ComboForecastMetrics, ComboForecastResponse, ComboForecastRiskLevel, + ProviderAutopilotReport, ComboHealthMetrics, ComboHealthResponse, ComboRecord, @@ -34,6 +35,7 @@ export interface ComboHealthAutopilotOptions { combos?: ComboRecord[]; healthResponse?: ComboHealthResponse; forecastResponse?: ComboForecastResponse; + providerHealthResponse?: ProviderAutopilotReport; } type ProviderIssueView = { @@ -103,7 +105,12 @@ function actionSet( case "open_combo_editor": return action(type, "Open combo editor", target, "/dashboard/combos"); case "run_combo_test": - return action(type, "Run combo test", target, "/dashboard/combos"); + return action( + type, + "Run combo test", + target, + `/dashboard/combos?test=${encodeURIComponent(target.comboId)}` + ); case "open_provider_health_autopilot": return action(type, "Open provider autopilot", target, "/dashboard/health"); case "review_quota_limits": @@ -447,7 +454,8 @@ export async function buildComboHealthAutopilotReport( now: options.now, combos: combosSnapshot, }), - buildProviderHealthAutopilotReport({ includeHealthy: false, includeActions: false }), + options.providerHealthResponse ?? + buildProviderHealthAutopilotReport({ includeHealthy: false, includeActions: false }), ]); const forecastsByComboId = new Map(forecast.combos.map((entry) => [entry.comboId, entry])); @@ -470,7 +478,7 @@ export async function buildComboHealthAutopilotReport( const degradedCount = allCombos.filter((combo) => combo.state === "degraded").length; const healthyCount = allCombos.filter((combo) => combo.state === "healthy").length; const issueCount = allCombos.reduce((sum, combo) => sum + combo.issues.length, 0); - const actionableCount = allCombos.reduce( + const suggestionCount = allCombos.reduce( (sum, combo) => sum + combo.issues.reduce((issueSum, issue) => issueSum + issue.actions.length, 0), 0 @@ -487,7 +495,8 @@ export async function buildComboHealthAutopilotReport( degradedCount, downCount, issueCount, - actionableCount, + suggestionCount, + actionableCount: suggestionCount, }, combos, }; diff --git a/src/lib/providers/validation/webProvidersA.ts b/src/lib/providers/validation/webProvidersA.ts index a1b20f1df9bd..52e9aa4c0ef1 100644 --- a/src/lib/providers/validation/webProvidersA.ts +++ b/src/lib/providers/validation/webProvidersA.ts @@ -13,11 +13,14 @@ import { normalizeSessionCookieHeader, } from "@/lib/providers/webCookieAuth"; -// kimi-web uses the international `www.kimi.com` Connect-RPC API. The legacy -// `kimi.moonshot.cn` domain now 307-redirects every non-CN visitor, and even -// if you bypass the redirect the old `/api/chat` REST endpoint is gone. The -// SPA exposes a profile probe at `GET /api/user` that returns the user object -// at the top level when the `Authorization: Bearer ` header is valid. +// kimi-web uses the international (west-facing) `www.kimi.ai` Connect-RPC API by +// default. `www.kimi.com` is the China-region endpoint — it serves China users but +// the China region is not reliably reachable from outside CN, so it is not the +// default. The legacy `kimi.moonshot.cn` domain now 307-redirects every non-CN +// visitor, and even if you bypass the redirect the old `/api/chat` REST endpoint is +// gone. The SPA exposes a profile probe at `GET /api/user` that returns the user +// object at the top level when the `Authorization: Bearer ` header is +// valid. Override the endpoint with KIMI_WEB_BASE_URL (opt-in). export async function validateKimiWebProvider({ apiKey }: any) { const rawCred = String(apiKey ?? "").trim(); if (!rawCred) { diff --git a/src/lib/quota/redisQuotaStore.ts b/src/lib/quota/redisQuotaStore.ts index 7c0e99ceb7ad..d9d17309eda7 100644 --- a/src/lib/quota/redisQuotaStore.ts +++ b/src/lib/quota/redisQuotaStore.ts @@ -72,7 +72,7 @@ export function resetRedisClient(): void { // Key helpers // --------------------------------------------------------------------------- -const KEY_PREFIX = "omniroute:quota"; +const KEY_PREFIX = `${process.env.REDIS_KEY_PREFIX?.trim() || "omniroute:"}quota`; function bucketKey(apiKeyId: string, dimensionKey: string, bucketIndex: number): string { return `${KEY_PREFIX}:${apiKeyId}:${dimensionKey}:${bucketIndex}`; diff --git a/src/lib/usage/internalUsageCommand.ts b/src/lib/usage/internalUsageCommand.ts index 257b8f9c1580..37f5f009d1c2 100644 --- a/src/lib/usage/internalUsageCommand.ts +++ b/src/lib/usage/internalUsageCommand.ts @@ -13,7 +13,7 @@ const TEXT_PLAIN_HEADERS = { "Content-Type": "text/plain; charset=utf-8" } as co type JsonRecord = Record; -interface UsageCommandApiKeyMetadata { +export interface UsageCommandApiKeyMetadata { id: string; name?: string; allowedConnections?: string[] | null; @@ -31,7 +31,7 @@ interface ProviderConnectionLike { quotaWindowThresholds?: Record | null; } -interface UsageSnapshot { +export interface UsageSnapshot { connectionId: string; provider: string; plan: unknown; @@ -39,7 +39,7 @@ interface UsageSnapshot { quotaWindowThresholds?: Record | null; } -interface UsageCommandSelection { +export interface UsageCommandSelection { preferredProvider?: string | null; preferredConnectionId?: string | null; } @@ -258,7 +258,7 @@ function snapshotFromConnection( }; } -async function collectUsageSnapshots( +export async function collectUsageSnapshots( metadata: UsageCommandApiKeyMetadata, deps: RequiredDeps ): Promise { @@ -525,6 +525,55 @@ function appendQuotaBlock( lines.push(`⏱ reset in ${formatResetIn(getResetAt(match?.quota ?? null), now)}`); } +/** + * Structured form of the usage command — what {@link buildUsageCommandText} + * renders as text, exposed as data for API consumers (the OmniCopilot panel + * asks for it via `?format=json`). Text and JSON share the exact same + * collectors, so the two can never disagree about a number. + * + * The key design constraint is the 403 case: a key without `allowUsageCommand` + * must reach the client as a *structured* reason, not a bare text error — a + * caller rendering a usage panel has to be able to tell "the server does not + * know your limits yet" apart from "this key may not ask". + */ +/** Discriminated so the caller never reads a data field off a refusal: + * `allowed:false` carries only `error`; `allowed:true` carries the data. */ +export type UsageCommandJson = + | { allowed: false; error: { message: string } } + | { + allowed: true; + /** Present only when the key opted into per-key usage limits. */ + personal: unknown | null; + /** The selected provider snapshot, or null when nothing is cached. */ + provider: UsageSnapshot | null; + /** Every connection's snapshot, so a panel can render Codex / Claude / + * OpenCode side by side instead of only the selected one (#11191). The + * single-pick in `provider` is a presentation choice for a terminal; the + * collector already gathered all of them. */ + providers: UsageSnapshot[]; + }; + +export async function buildUsageCommandJson( + metadata: UsageCommandApiKeyMetadata, + deps: InternalUsageCommandDeps = {}, + selection: UsageCommandSelection = {} +): Promise { + const resolvedDeps = await normalizeDeps(deps); + const personal = + metadata.usageLimitEnabled === true + ? await resolvedDeps.getApiKeyUsageLimitStatus( + { + ...metadata, + preferredProvider: selection.preferredProvider ?? metadata.preferredProvider ?? null, + }, + { now: resolvedDeps.now } + ) + : null; + const snapshots = await collectUsageSnapshots(metadata, resolvedDeps); + const provider = selectUsageSnapshot(snapshots, selection); + return { allowed: true, personal, provider, providers: snapshots }; +} + export async function buildUsageCommandText( metadata: UsageCommandApiKeyMetadata, deps: InternalUsageCommandDeps = {}, @@ -588,6 +637,17 @@ function inferHttpUsageCommandSelection(request: Request): UsageCommandSelection } } +/** `?format=json` (or `?format=JSON`) — anything else falls back to the text + * form, which is the historical contract of this endpoint. */ +function wantsUsageCommandJson(request: Request): boolean { + try { + const format = new URL(request.url, "http://localhost").searchParams.get("format"); + return format !== null && format.trim().toLowerCase() === "json"; + } catch { + return false; + } +} + function createPlainUsageCommandResponse(text: string, status = 200): Response { return new Response(text, { status, headers: TEXT_PLAIN_HEADERS }); } @@ -764,22 +824,45 @@ export async function handleInternalUsageCommandHttpRequest( ): Promise { try { const resolvedDeps = await normalizeDeps(deps); + const json = wantsUsageCommandJson(request); const apiKey = extractUsageCommandApiKey(request); if (!apiKey || !(await resolvedDeps.isValidApiKey(apiKey))) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } const metadata = await resolvedDeps.getApiKeyMetadata(apiKey); if (!metadata?.id) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } if (metadata.allowUsageCommand !== true) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_DISABLED_MESSAGE } } satisfies UsageCommandJson, + { status: 403 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_DISABLED_MESSAGE, 403); } + const selection = inferHttpUsageCommandSelection(request); + if (json) { + return Response.json(await buildUsageCommandJson(metadata, resolvedDeps, selection)); + } return createPlainUsageCommandResponse( - await buildUsageCommandText(metadata, resolvedDeps, inferHttpUsageCommandSelection(request)) + await buildUsageCommandText(metadata, resolvedDeps, selection) ); } catch (err) { const body = buildErrorBody(500, err instanceof Error ? err.message : String(err)); diff --git a/src/server/authz/pipeline.ts b/src/server/authz/pipeline.ts index 9f4e46bed1eb..d8dacf376d7e 100644 --- a/src/server/authz/pipeline.ts +++ b/src/server/authz/pipeline.ts @@ -205,6 +205,7 @@ function drainingResponse(requestId: string): NextResponse { { status: 503 } ); response.headers.set(AUTHZ_HEADER_REQUEST_ID, requestId); + response.headers.set("Retry-After", "5"); return response; } diff --git a/src/shared/components/ProviderIcon.tsx b/src/shared/components/ProviderIcon.tsx index aeb6f777ba03..56d54bf0dc27 100644 --- a/src/shared/components/ProviderIcon.tsx +++ b/src/shared/components/ProviderIcon.tsx @@ -263,6 +263,7 @@ const KNOWN_PNGS = new Set([ "linkup-search", "llamafile", "llamagate", + "logfare", "maritalk", "nanobot", "nanogpt", diff --git a/src/shared/constants/providers.ts b/src/shared/constants/providers.ts index 23f9c53b39ce..b9b2b6951dbd 100644 --- a/src/shared/constants/providers.ts +++ b/src/shared/constants/providers.ts @@ -106,7 +106,6 @@ export const AGGREGATOR_PROVIDER_IDS = new Set([ "empower", "poe", "chutes", - "hackclub", "freetheai", "g4f-groq", "g4f-gemini", @@ -142,6 +141,7 @@ export const AGGREGATOR_PROVIDER_IDS = new Set([ "void-ai", "helixmind", "tabitoken", + "logfare", ]); export const ENTERPRISE_CLOUD_PROVIDER_IDS = new Set([ @@ -237,9 +237,7 @@ export function isSelfHostedChatProvider(providerId: unknown): boolean { const EXPLICIT_OPTIONAL_APIKEY_PROVIDER_IDS = new Set([ "searxng-search", "firecrawl", - "pollinations", "copilot-web", - "hackclub", "g4f-groq", "g4f-gemini", "g4f-pollinations", diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts index 2baf13d3aaed..0b8d07759f3a 100644 --- a/src/shared/constants/providers/apikey/gateways.ts +++ b/src/shared/constants/providers/apikey/gateways.ts @@ -1265,6 +1265,29 @@ export const APIKEY_PROVIDERS_GATEWAYS = { apiHint: "Create a helix- key and use https://helixmind.online/v1. OpenAI requests use Bearer authentication; the Anthropic-compatible messages endpoint accepts x-api-key.", }, + // Logfare (https://logfare.ai) — free OpenAI-compatible inference, live-verified + // 2026-08-21 (real /v1/models catalog; 11 chat-capable models incl. kimi-k3, + // deepseek-v4-pro, glm-5.2, gpt-5.6-luna). Key issued instantly at /register + // (username/password, no email). ⚠️ Logfare logs every request in exchange for + // free inference (opt out at /consent) — surfaced in freeNote per the catalog + // convention for data-collecting free providers. + logfare: { + id: "logfare", + alias: "logfare", + name: "Logfare", + icon: "auto_awesome", + color: "#22C55E", + textIcon: "LF", + website: "https://logfare.ai", + hasFree: true, + freeNote: + "Free OpenAI-compatible inference — no rate limits, no card. Logfare logs every request (prompts, completions, metadata) for internal research; opt out at /consent. Read https://logfare.ai/tos and https://logfare.ai/privacy before use.", + authHint: + "Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token.", + apiHint: + "Create a free API key at https://logfare.ai/register, then use https://logfare.ai/v1 as the OpenAI-compatible base URL. Note the request-logging policy: prompts, completions and metadata are logged for research (opt out at https://logfare.ai/consent).", + passthroughModels: true, + }, // TabiToken (https://tabitoken.com) — NewAPI-based Claude gateway. Its public pricing // endpoint lists a Claude-only catalog (Opus 5 / 4.8, each with a -thinking variant), // every model accepting the Anthropic and OpenAI protocols. diff --git a/src/shared/constants/providers/web-cookie.ts b/src/shared/constants/providers/web-cookie.ts index 72d1440d1f5e..65e63ac615d3 100644 --- a/src/shared/constants/providers/web-cookie.ts +++ b/src/shared/constants/providers/web-cookie.ts @@ -323,14 +323,9 @@ export const WEB_COOKIE_PROVIDERS = { icon: "auto_awesome", color: "#2563EB", textIcon: "KW", - // Kimi official-partnership aff link (2026-07) — the "Kimi Coding Plan" - // tracking link (same origin as the plain www.kimi.com login flow below, - // so the "Open {host}" credential guide in WebSessionCredentialGuide.tsx / - // AddApiKeyModal.tsx is unaffected: origin, not path, decides localStorage - // access). Was `https://www.kimi.com` (no aff attribution). - website: "https://www.kimi.com/code?aff=omniroute", + website: "https://www.kimi.ai", authHint: - "Paste access_token from www.kimi.com DevTools → Application → Local Storage. A legacy kimi-auth cookie is also accepted.", + "Paste access_token from www.kimi.ai DevTools → Application → Local Storage. A legacy kimi-auth cookie is also accepted.", subscriptionRisk: true, riskNoticeVariant: "webCookie", }, diff --git a/src/shared/middleware/chatBodyAdmission.ts b/src/shared/middleware/chatBodyAdmission.ts index 1c3c25b904b8..9b20d78e5aea 100644 --- a/src/shared/middleware/chatBodyAdmission.ts +++ b/src/shared/middleware/chatBodyAdmission.ts @@ -18,6 +18,7 @@ import { CORS_HEADERS } from "../utils/cors"; import { createHmac } from "crypto"; import v8 from "node:v8"; +import { trackRequest } from "../../lib/gracefulShutdown"; function parsePositiveInt(value: string | undefined, fallback: number): number { const parsed = Number.parseInt(String(value), 10); @@ -229,6 +230,7 @@ export class ChatAdmissionController { tryAcquireHealthyHeadroom(): ChatAdmissionLease | null { if (this.#activeHealthy >= this.healthyHeadroom) return null; this.#activeHealthy += 1; + const done = trackRequest(); let released = false; return { get released() { @@ -238,6 +240,7 @@ export class ChatAdmissionController { if (released) return; released = true; this.#activeHealthy = Math.max(0, this.#activeHealthy - 1); + done(); }, }; } @@ -264,6 +267,7 @@ export class ChatAdmissionController { tryAcquireHeavy(): ChatAdmissionLease | null { if (this.#activeHeavy >= this.maxHeavyInFlight) return null; this.#activeHeavy += 1; + const done = trackRequest(); let released = false; return { get released() { @@ -273,6 +277,7 @@ export class ChatAdmissionController { if (released) return; released = true; this.#activeHeavy = Math.max(0, this.#activeHeavy - 1); + done(); this.#dispatchFair(); }, }; diff --git a/src/shared/network/outboundUrlGuard.ts b/src/shared/network/outboundUrlGuard.ts index e7175da0bc54..45a7ebde7bed 100644 --- a/src/shared/network/outboundUrlGuard.ts +++ b/src/shared/network/outboundUrlGuard.ts @@ -1,4 +1,10 @@ -import { isIP } from "node:net"; +import { ipVersion, isPrivateHost, normalizeHost } from "./privateHost"; + +// #11122: the host classification lives in `./privateHost.ts` because +// `open-sse/config/providerRegistry.ts` imports it from a module reachable by a browser +// bundle, and `node:net` cannot be resolved there. Re-exported so every existing caller of +// `isPrivateHost` from this module keeps working unchanged. +export { isPrivateHost }; export const PROVIDER_URL_BLOCKED_MESSAGE = "Blocked private or local provider URL"; export const CLOUD_METADATA_BLOCKED_MESSAGE = "Blocked cloud-metadata endpoint"; @@ -29,61 +35,6 @@ export class OutboundUrlGuardError extends Error { } } -function normalizeHost(hostname: string) { - const normalized = hostname.trim().toLowerCase(); - if (normalized.startsWith("[") && normalized.endsWith("]")) { - return normalized.slice(1, -1); - } - return normalized; -} - -export function isPrivateHost(hostname: string) { - const normalized = normalizeHost(hostname); - if (!normalized) return true; - - if ( - normalized === "localhost" || - normalized === "0.0.0.0" || - // `::` is the IPv6 twin of `0.0.0.0`: connecting to it reaches a service bound - // to the IPv6 loopback, so it has to be refused alongside its IPv4 spelling. - normalized === "::" || - normalized === "127.0.0.1" || - normalized === "::1" || - normalized.endsWith(".localhost") || - normalized.endsWith(".local") || - // `.internal` is reserved for private use (ICANN-style) and is the - // hostname suffix used by GCP/Azure metadata probes - // (e.g. `metadata.google.internal`). - normalized.endsWith(".internal") || - normalized.startsWith("::ffff:") - ) { - return true; - } - - if (isIP(normalized) === 4) { - const octets = normalized.split(".").map((segment) => parseInt(segment, 10)); - const [a, b] = octets; - - if (a === 0 || a === 10 || a === 127) return true; - if (a === 169 && b === 254) return true; - if (a === 192 && b === 168) return true; - if (a === 172 && b >= 16 && b <= 31) return true; - if (a === 100 && b >= 64 && b <= 127) return true; - return false; - } - - if (isIP(normalized) === 6) { - return ( - normalized === "::1" || - normalized.startsWith("fc") || - normalized.startsWith("fd") || - normalized.startsWith("fe80:") - ); - } - - return false; -} - // WHATWG URL serialises an IPv4-mapped IPv6 address as hextets, so // `http://[::ffff:169.254.169.254]/` reaches these helpers as `::ffff:a9fe:a9fe`. // Matching the dotted spelling alone therefore misses every mapped address that @@ -92,7 +43,7 @@ function mappedIpv4Host(hostname: string): string | null { const normalized = normalizeHost(hostname); if (!normalized.startsWith("::ffff:")) return null; const embedded = normalized.slice("::ffff:".length); - if (isIP(embedded) === 4) return embedded; + if (ipVersion(embedded) === 4) return embedded; const hextets = embedded.split(":"); if (hextets.length !== 2) return null; const [high, low] = hextets.map((part) => @@ -202,3 +153,4 @@ export function parseAndValidateNonMetadataUrl(input: string | URL) { // opencode.ts) where no `tsconfig.json` is present to resolve the `@/*` path alias. Keeping // this module free of ANY `@/`-aliased import is what makes it safe to load from the CLI. // Do not add a `@/`-aliased import here — see docs/security/… (packaging) and #7682. +// The same rule binds `./privateHost.ts`, which this module re-exports from. diff --git a/src/shared/network/privateHost.ts b/src/shared/network/privateHost.ts new file mode 100644 index 000000000000..f3120e3d1bfd --- /dev/null +++ b/src/shared/network/privateHost.ts @@ -0,0 +1,103 @@ +// Host classification shared by the outbound URL guard and the provider registry. +// +// #11122: `open-sse/config/providerRegistry.ts` needs `isPrivateHost`, and that module is +// reachable from `ProviderDetailPageClient.tsx`. `outboundUrlGuard.ts` reached for `node:net`'s +// `isIP`, so importing it from the registry broke the browser bundle with +// `Could not resolve "node:net"` (caught by tests/unit/media-page-client-browser-bundle.test.ts, +// which has been red on the release branch since #11122 merged). +// The classification therefore lives here, on a pure-JS `ipVersion`, with NO platform imports. +// +// Two constraints this module MUST keep — both enforced by tests: +// 1. No `node:*` import: it is bundled for the browser. +// 2. No `@/`-aliased import: `./outboundUrlGuard.ts` re-exports from here and is loaded by the +// packaged CLI (`omniroute setup-opencode`), where no tsconfig resolves the alias (#7682). + +// Vendored from Node's own `lib/internal/net.js` so `ipVersion` stays verdict-for-verdict +// identical to `isIP` — a NARROWER match would silently reclassify a private address as public +// and open the very egress the guard exists to close. `tests/unit/private-host-ip-parity-11122` +// asserts that parity against `node:net` directly. +const V4_SEG = "(?:25[0-5]|2[0-4]\\d|1\\d\\d|[1-9]?\\d)"; +const V4_STR = `(?:${V4_SEG}\\.){3}${V4_SEG}`; +const V6_SEG = "(?:[0-9a-fA-F]{1,4})"; + +const IPV4_RE = new RegExp(`^${V4_STR}$`); + +const IPV6_RE = new RegExp( + "^(?:" + + `(?:${V6_SEG}:){7}(?:${V6_SEG}|:)|` + + `(?:${V6_SEG}:){6}(?:${V4_STR}|:${V6_SEG}|:)|` + + `(?:${V6_SEG}:){5}(?::${V4_STR}|(?::${V6_SEG}){1,2}|:)|` + + `(?:${V6_SEG}:){4}(?:(?::${V6_SEG}){0,1}:${V4_STR}|(?::${V6_SEG}){1,3}|:)|` + + `(?:${V6_SEG}:){3}(?:(?::${V6_SEG}){0,2}:${V4_STR}|(?::${V6_SEG}){1,4}|:)|` + + `(?:${V6_SEG}:){2}(?:(?::${V6_SEG}){0,3}:${V4_STR}|(?::${V6_SEG}){1,5}|:)|` + + `(?:${V6_SEG}:){1}(?:(?::${V6_SEG}){0,4}:${V4_STR}|(?::${V6_SEG}){1,6}|:)|` + + `(?::(?:(?::${V6_SEG}){0,5}:${V4_STR}|(?::${V6_SEG}){1,7}|:))` + + ")(?:%[0-9a-zA-Z-.:]{1,64})?$" +); + +// Longest legal literal is 45 chars (`ffff:…:255.255.255.255`) plus a `%zone`. Every quantifier +// above is bounded, and this guard keeps the alternation from ever seeing a long hostile string +// (AGENTS.md → "Regex Security (ReDoS)"). +const MAX_IP_LITERAL_LENGTH = 110; + +/** Pure-JS `node:net#isIP`: 4, 6, or 0 when the string is not an IP literal. */ +export function ipVersion(host: string): 0 | 4 | 6 { + if (!host || host.length > MAX_IP_LITERAL_LENGTH) return 0; + if (IPV4_RE.test(host)) return 4; + return IPV6_RE.test(host) ? 6 : 0; +} + +export function normalizeHost(hostname: string) { + const normalized = hostname.trim().toLowerCase(); + if (normalized.startsWith("[") && normalized.endsWith("]")) { + return normalized.slice(1, -1); + } + return normalized; +} + +export function isPrivateHost(hostname: string) { + const normalized = normalizeHost(hostname); + if (!normalized) return true; + + if ( + normalized === "localhost" || + normalized === "0.0.0.0" || + // `::` is the IPv6 twin of `0.0.0.0`: connecting to it reaches a service bound + // to the IPv6 loopback, so it has to be refused alongside its IPv4 spelling. + normalized === "::" || + normalized === "127.0.0.1" || + normalized === "::1" || + normalized.endsWith(".localhost") || + normalized.endsWith(".local") || + // `.internal` is reserved for private use (ICANN-style) and is the + // hostname suffix used by GCP/Azure metadata probes + // (e.g. `metadata.google.internal`). + normalized.endsWith(".internal") || + normalized.startsWith("::ffff:") + ) { + return true; + } + + if (ipVersion(normalized) === 4) { + const octets = normalized.split(".").map((segment) => parseInt(segment, 10)); + const [a, b] = octets; + + if (a === 0 || a === 10 || a === 127) return true; + if (a === 169 && b === 254) return true; + if (a === 192 && b === 168) return true; + if (a === 172 && b >= 16 && b <= 31) return true; + if (a === 100 && b >= 64 && b <= 127) return true; + return false; + } + + if (ipVersion(normalized) === 6) { + return ( + normalized === "::1" || + normalized.startsWith("fc") || + normalized.startsWith("fd") || + normalized.startsWith("fe80:") + ); + } + + return false; +} diff --git a/src/shared/services/opencodeConfig.ts b/src/shared/services/opencodeConfig.ts index a4bbd6559120..a91ca5069c5c 100644 --- a/src/shared/services/opencodeConfig.ts +++ b/src/shared/services/opencodeConfig.ts @@ -84,11 +84,29 @@ export const buildOpenCodeProviderConfig = ({ }; }; +export const buildOpenCodeV2ProviderConfig = ( + input: OpenCodeConfigInput +): Record => { + const v1Config = buildOpenCodeProviderConfig(input); + return { + name: v1Config.name, + package: "@opencode-ai/ai/providers/openai-compatible", + settings: { + baseURL: v1Config.options.baseURL, + apiKey: v1Config.options.apiKey, + }, + models: v1Config.models, + }; +}; + export const buildOpenCodeConfigDocument = (input: OpenCodeConfigInput) => ({ $schema: "https://opencode.ai/config.json", provider: { omniroute: buildOpenCodeProviderConfig(input), }, + providers: { + omniroute: buildOpenCodeV2ProviderConfig(input), + }, }); export const mergeOpenCodeConfig = ( @@ -100,18 +118,18 @@ export const mergeOpenCodeConfig = ( ? existingConfig : {}; - // Same guard as the root above, one level down. Spreading a non-object here - // does not throw, it splays the value into index keys: an existing - // `"provider": ["a", "b"]` merged to `{"0": "a", "1": "b", omniroute: ... }` - // and a string was exploded one character per key. mergeOpenCodeConfigText - // refuses the same input outright, so the two disagreed on what to do with a - // malformed config. const existingProvider = (safeConfig as Record).provider; const safeProvider = existingProvider && typeof existingProvider === "object" && !Array.isArray(existingProvider) ? (existingProvider as Record) : {}; + const existingProviders = (safeConfig as Record).providers; + const safeProviders = + existingProviders && typeof existingProviders === "object" && !Array.isArray(existingProviders) + ? (existingProviders as Record) + : {}; + return { ...safeConfig, $schema: safeConfig.$schema || "https://opencode.ai/config.json", @@ -119,6 +137,10 @@ export const mergeOpenCodeConfig = ( ...safeProvider, omniroute: buildOpenCodeProviderConfig(input), }, + providers: { + ...safeProviders, + omniroute: buildOpenCodeV2ProviderConfig(input), + }, }; }; @@ -127,6 +149,7 @@ export const mergeOpenCodeConfigText = ( input: OpenCodeConfigInput ) => { const providerConfig = buildOpenCodeProviderConfig(input); + const v2ProviderConfig = buildOpenCodeV2ProviderConfig(input); const content = typeof existingText === "string" ? existingText : ""; const trimmedContent = content.trim(); @@ -161,6 +184,11 @@ export const mergeOpenCodeConfigText = ( const providerEdits = modify(nextText, ["provider", "omniroute"], providerConfig, { formattingOptions: { insertSpaces: true, tabSize: 2 }, }); + nextText = applyEdits(nextText, providerEdits); + + const v2ProviderEdits = modify(nextText, ["providers", "omniroute"], v2ProviderConfig, { + formattingOptions: { insertSpaces: true, tabSize: 2 }, + }); - return applyEdits(nextText, providerEdits); + return applyEdits(nextText, v2ProviderEdits); }; diff --git a/src/shared/types/utilization.ts b/src/shared/types/utilization.ts index 36f0d7dabcf3..0a1e9dc900d6 100644 --- a/src/shared/types/utilization.ts +++ b/src/shared/types/utilization.ts @@ -260,7 +260,9 @@ export interface ComboAutopilotReport { degradedCount: number; downCount: number; issueCount: number; - actionableCount: number; + suggestionCount: number; + /** @deprecated Use suggestionCount instead. Kept as an alias for backward compatibility; remove after 2 releases. */ + actionableCount?: number; }; combos: ComboAutopilotCombo[]; } diff --git a/src/shared/utils/noAuthProviders.ts b/src/shared/utils/noAuthProviders.ts index ac025be9f868..83b3235da7ec 100644 --- a/src/shared/utils/noAuthProviders.ts +++ b/src/shared/utils/noAuthProviders.ts @@ -22,8 +22,10 @@ export function isProviderBlockedByIdOrAlias( ): boolean { const blockedProviderSet = normalizeBlockedProviderSet(blockedProviders); const provider = getProviderById(providerId) as ProviderWithAlias | undefined; + const baseId = providerId.replace(/-search$/, ""); return ( blockedProviderSet.has(providerId) || + blockedProviderSet.has(baseId) || (typeof provider?.alias === "string" && blockedProviderSet.has(provider.alias)) ); } diff --git a/src/shared/utils/rateLimiter.ts b/src/shared/utils/rateLimiter.ts index bb268ba5f039..7b36a2173697 100644 --- a/src/shared/utils/rateLimiter.ts +++ b/src/shared/utils/rateLimiter.ts @@ -3,6 +3,10 @@ import type Redis from "ioredis"; // Redis is optional. When REDIS_URL is unset, use a process-local fallback // instead of probing localhost on every API request. const REDIS_URL = process.env.REDIS_URL?.trim() || ""; + +// Namespace prefix for all OmniRoute Redis keys. Prevents key collisions when +// OmniRoute shares a Redis instance with other apps (e.g. on 127.0.0.1:6379). +const REDIS_KEY_PREFIX = process.env.REDIS_KEY_PREFIX?.trim() || "omniroute:"; if (process.env.NODE_ENV === "production" && !REDIS_URL) { console.warn("[REDIS] REDIS_URL is not set in production. Using in-memory rate limiting."); } @@ -72,6 +76,7 @@ export function getRedisClient(): Promise { const client = new RedisCtor(REDIS_URL, { maxRetriesPerRequest: 3, enableReadyCheck: false, + keyPrefix: REDIS_KEY_PREFIX, retryStrategy(times) { return Math.min(times * 50, 2000); // Exponential backoff }, diff --git a/src/shared/validation/helpers.ts b/src/shared/validation/helpers.ts index 4e486c7287d3..c811d6d4bf02 100644 --- a/src/shared/validation/helpers.ts +++ b/src/shared/validation/helpers.ts @@ -4,6 +4,10 @@ import { z } from "zod"; type ValidationErrorDetail = { field: string; message: string; + // Present for `unrecognized_keys` issues: the unknown key names that were + // refused (e.g. a typo'd override key). Surfaced by callers so clients can + // tell exactly which keys were rejected. + keys?: string[]; }; type ValidationErrorPayload = { @@ -45,6 +49,9 @@ export function validateBody( details: issues.map((e) => ({ field: e.path.join("."), message: e.message, + ...(("keys" in e && (e as { keys?: string[] }).keys) + ? { keys: (e as { keys: string[] }).keys } + : {}), })), }, }; diff --git a/src/shared/validation/schemas/combo.ts b/src/shared/validation/schemas/combo.ts index edeca47e729a..db825e11cc9d 100644 --- a/src/shared/validation/schemas/combo.ts +++ b/src/shared/validation/schemas/combo.ts @@ -321,7 +321,7 @@ export const createComboSchema = z .object({ name: comboNameSchema, description: z.string().max(2000).optional(), - models: z.array(comboModelEntry).optional().default([]), + models: z.array(comboModelEntry).min(1, "a combo requires at least one model"), strategy: comboStrategySchema.optional().default("priority"), config: comboRuntimeConfigSchema.optional(), allowedProviders: z.array(z.string().trim().min(1).max(200)).max(100).optional(), @@ -380,8 +380,9 @@ export const updateComboSchema = z .object({ name: comboNameSchema.optional(), description: z.string().max(2000).optional().nullable(), - // Creation may leave `models` empty (`omniroute combo create` drafts one - // that way); an update may not, or a working combo loses every target. + // An update may not remove every model from a combo, or a working combo + // loses every target. Creation refuses an empty list too: since the CLI + // gained --models (#10954), an empty draft has no remaining legitimate path. models: z .array(comboModelEntry) .min(1, "an update cannot remove every model from a combo") diff --git a/src/shared/validation/schemas/provider.ts b/src/shared/validation/schemas/provider.ts index b3272f70fb64..4cfe7afff453 100644 --- a/src/shared/validation/schemas/provider.ts +++ b/src/shared/validation/schemas/provider.ts @@ -420,6 +420,25 @@ export const providerNodeValidateSchema = z.object({ modelId: z.string().trim().max(200).optional().or(z.literal("")), }); +// rate-limit override numeric fields must reject operator intent loss. +// `z.coerce.number()` silently turns "" into 0 and "60abc" into NaN, which +// would drop or distort the value instead of rejecting it. Preprocess first so +// an empty/non-numeric string fails validation (surfaced as a 400), while still +// coercing legit numeric strings like "60". +function rateLimitOverrideNumber(max: number) { + return z.preprocess( + (raw) => { + if (typeof raw === "string") { + if (raw.trim() === "") return NaN; + const parsed = Number(raw); + return Number.isNaN(parsed) ? raw : parsed; + } + return raw; + }, + z.coerce.number().int().min(0).max(max) + ); +} + export const updateProviderConnectionSchema = z .object({ name: z.string().max(200).optional(), @@ -468,17 +487,24 @@ export const updateProviderConnectionSchema = z projectId: z.union([z.string(), z.null()]).optional(), // Per-connection rate limit overrides — overrides the global RequestQueueSettings // for this connection. Set to null to clear all overrides. + // Per-connection rate limit overrides — overrides the global + // RequestQueueSettings for this connection. Set to null to clear all + // overrides. `.strict()` rejects unknown keys (e.g. a typo'd `tmp`) with a + // 400 instead of silently stripping them: the operator's intent is + // never dropped without an error. `.nullable()` (rather than a + // `z.union([z.null(), …])`) keeps the `unrecognized_keys` issue at the top + // level so the rejected key name survives into the 400 response. rateLimitOverrides: z - .union([ - z.null(), - z.object({ - rpm: z.coerce.number().int().min(0).max(1_000_000).optional(), - tpm: z.coerce.number().int().min(0).max(100_000_000).optional(), - tpd: z.coerce.number().int().min(0).max(10_000_000_000).optional(), - minTime: z.coerce.number().int().min(0).max(60_000).optional(), - maxConcurrent: z.coerce.number().int().min(0).max(10_000).optional(), - }), - ]) + .object({ + rpm: rateLimitOverrideNumber(1_000_000).optional(), + tpm: rateLimitOverrideNumber(100_000_000).optional(), + tpd: rateLimitOverrideNumber(10_000_000_000).optional(), + minTime: rateLimitOverrideNumber(60_000).optional(), + maxConcurrent: rateLimitOverrideNumber(10_000).optional(), + }) + .partial() + .strict() + .nullable() .optional(), proxyEnabled: z.boolean().optional(), perKeyProxyEnabled: z.boolean().optional(), diff --git a/src/shared/validation/settingsSchemas.ts b/src/shared/validation/settingsSchemas.ts index 388e3f73c74f..beefb5fc8f32 100644 --- a/src/shared/validation/settingsSchemas.ts +++ b/src/shared/validation/settingsSchemas.ts @@ -259,6 +259,48 @@ export const updateSettingsSchema = z.object({ }) ) .optional(), + /** + * Operator-declared per-provider error rules. Consulted BEFORE the built-in + * `providerRuleRegistry` in open-sse/config/providerErrorRules.ts so an + * operator can add a scope/cooldown/reason override for a provider without + * editing the catalog. Matches are plain case-insensitive SUBSTRINGS of the + * error body (never RegExp) to keep the classification hot path ReDoS-safe. + * Bounded to 50 rules total so a misconfigured setting cannot blow up the + * matcher. + */ + providerErrorRules: z + .record( + z.string().trim().min(1).max(100), + z.array( + z.object({ + status: z.number().int().min(100).max(599), + match: z.string().min(1).max(200), + scope: z.enum(["model", "provider", "connection"]), + reason: z + .enum([ + "auth_error", + "quota_exhausted", + "rate_limit_exceeded", + "model_capacity", + "server_error", + "unknown", + ]) + .optional(), + cooldownMs: z.number().int().min(0).max(86_400_000).optional(), + }) + ) + ) + .optional() + .superRefine((value, ctx) => { + if (!value) return; + const total = Object.values(value).reduce((n, rules) => n + rules.length, 0); + if (total > 50) { + ctx.addIssue({ + code: z.ZodIssueCode.custom, + message: `providerErrorRules: at most 50 rules total, got ${total}`, + }); + } + }), // #6168: global session-stickiness opt-out (per-combo config overrides this). disableSessionStickiness: z.boolean().optional(), /** Keep eligible combo targets close to the provider-side prompt cache. */ diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 3ec6e6c6d6c2..49a29c1b69f4 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -1851,7 +1851,7 @@ async function handleSingleModelChat( modelPinned: runtimeOptions?.modelPinned ?? false, routingComboId: runtimeOptions?.routingComboId ?? null, sessionAffinityKey: runtimeOptions.sessionAffinityKey ?? null, - reasoningTransportFallback: runtimeOptions.reasoningTransportFallback ?? "skip", + reasoningTransportFallback: runtimeOptions.reasoningTransportFallback ?? "drop", managedLease: runtimeOptions.managedLease ?? null, }, runtimeOptions diff --git a/src/sse/handlers/chatHelpers.ts b/src/sse/handlers/chatHelpers.ts index 36bd4077408a..bcf9a51c62cd 100644 --- a/src/sse/handlers/chatHelpers.ts +++ b/src/sse/handlers/chatHelpers.ts @@ -422,7 +422,7 @@ export async function executeChatWithBreaker({ conversationId = null, modelPinned = false, routingComboId = null, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", sessionAffinityKey = null, managedLease = null, }: ExecuteChatWithBreakerOptions): Promise { diff --git a/src/types/databaseSettings.ts b/src/types/databaseSettings.ts index 6bcc210e5fd2..2b4820e19825 100644 --- a/src/types/databaseSettings.ts +++ b/src/types/databaseSettings.ts @@ -44,6 +44,7 @@ export interface DatabaseSettings { quotaSnapshots: number; compressionAnalytics: number; mcpAudit: number; + configAudit: number; a2aEvents: number; callLogs: number; usageHistory: number; @@ -114,6 +115,7 @@ export const DEFAULT_DATABASE_SETTINGS: Omit", - "Content-Type": "application/json" - }, - "nonStream": { - "Authorization": "Bearer ", - "Content-Type": "application/json" - }, - "oauth": { - "Accept": "text/event-stream", - "Authorization": "Bearer ", - "Content-Type": "application/json" - } - }, - "url": { - "nonStream": "https://ai.hackclub.com/proxy/v1/chat/completions", - "stream": "https://ai.hackclub.com/proxy/v1/chat/completions" - } - }, "hailuo-web": { "format": "openai", "headers": { @@ -2826,8 +2803,8 @@ } }, "url": { - "nonStream": "https://www.hailuo.ai", - "stream": "https://www.hailuo.ai" + "nonStream": "https://chat.minimax.io", + "stream": "https://chat.minimax.io" } }, "haiper": { @@ -3364,8 +3341,8 @@ } }, "url": { - "nonStream": "https://www.kimi.com", - "stream": "https://www.kimi.com" + "nonStream": "https://www.kimi.ai", + "stream": "https://www.kimi.ai" } }, "kiro": { @@ -3608,6 +3585,29 @@ "stream": "https://arena.ai/nextjs-api/stream/create-evaluation" } }, + "logfare": { + "format": "openai", + "headers": { + "apiKey": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "nonStream": { + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "oauth": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + } + }, + "url": { + "nonStream": "https://logfare.ai/v1/chat/completions", + "stream": "https://logfare.ai/v1/chat/completions" + } + }, "longcat": { "format": "openai", "headers": { diff --git a/tests/unit/account-fallback-service.test.ts b/tests/unit/account-fallback-service.test.ts index 3711e6217c80..266179ec7a53 100644 --- a/tests/unit/account-fallback-service.test.ts +++ b/tests/unit/account-fallback-service.test.ts @@ -453,6 +453,25 @@ test("hasPerModelQuota returns true for GitHub Copilot provider (#1624)", () => assert.equal(hasPerModelQuota("github", "gpt-5-mini"), true); }); +test("hasPerModelQuota honors shared-registry passthrough providers (#11071)", () => { + // These declare passthroughModels:true in src/shared/constants/providers/, but are absent + // from the open-sse REGISTRY passthrough set and are neither local nor self-hosted — so the + // isLocalProvider/isSelfHostedChatProvider branch (#11078) never reaches them. Without the + // shared-registry lookup a missing model on one of these cools the WHOLE connection. + assert.equal(hasPerModelQuota("novita"), true); + assert.equal(hasPerModelQuota("uncloseai"), true); + assert.equal(hasPerModelQuota("orcarouter"), true); + + // Already covered by the local/self-hosted branch — asserted so this port cannot regress it. + assert.equal(hasPerModelQuota("ollama-local"), true); + assert.equal(hasPerModelQuota("lm-studio"), true); + assert.equal(hasPerModelQuota("vllm"), true); + + // Neither declared in the shared registry nor local: a failure here is still connection-wide. + assert.equal(hasPerModelQuota("openai"), false); + assert.equal(hasPerModelQuota("anthropic"), false); +}); + test("Codex Spark 429s are scoped away from normal Codex models", () => { const connectionId = `codex-${Date.now()}`; clearModelLock("codex", connectionId, "gpt-5.3-codex-spark"); @@ -1659,7 +1678,11 @@ test("#10460: model-unsupported 400 handles various phrasings", async () => { // Verify connection stays healthy after each iteration const conn = await providersDb.getProviderConnectionById(id); assert.ok(!conn.rateLimitedUntil, `"${errorText}" must not rate-limit connection`); - assert.notStrictEqual(conn.testStatus, "unavailable", `"${errorText}" must not mark unavailable`); + assert.notStrictEqual( + conn.testStatus, + "unavailable", + `"${errorText}" must not mark unavailable` + ); } }); @@ -1692,7 +1715,11 @@ test("#10460: non-400 status with model-unsupported text does NOT trigger guard" "test-model" ); - assert.strictEqual(result.shouldFallback, true, "non-400 must not be short-circuited by model guard"); + assert.strictEqual( + result.shouldFallback, + true, + "non-400 must not be short-circuited by model guard" + ); // The key assertion: guard returns shouldFallback:false. If we get here with // shouldFallback:true, the guard did NOT fire (correct behavior). }); @@ -1723,7 +1750,11 @@ test("#10460: auth-credential 400 text does NOT match model-unsupported guard", // This text does NOT match MODEL_ACCESS_DENIED_PATTERNS (verified by regex test) // so it falls through to checkFallbackError which returns shouldFallback:false for generic 400 - assert.strictEqual(result.shouldFallback, false, "auth-credential 400 must not be caught by model guard"); + assert.strictEqual( + result.shouldFallback, + false, + "auth-credential 400 must not be caught by model guard" + ); // The generic 400 path returns cooldownMs:0 — same as the guard, but the // connection was NOT touched (no rateLimitedUntil set). This distinguishes // it from the normal fallback path which would set a cooldown. @@ -1737,7 +1768,11 @@ test("#10460: guard early return does not touch DB (distinguishes from normal pa // Guard path: model-unsupported 400 → shouldFallback:false, cooldownMs:0, no DB change const guardResult = await auth.markAccountUnavailable( - connId, 400, "The requested model is not supported", "github", "test-model" + connId, + 400, + "The requested model is not supported", + "github", + "test-model" ); assert.strictEqual(guardResult.shouldFallback, false); assert.strictEqual(guardResult.cooldownMs, 0); diff --git a/tests/unit/account-rotation.test.ts b/tests/unit/account-rotation.test.ts index f4864ee2dd84..a682f02a0d8a 100644 --- a/tests/unit/account-rotation.test.ts +++ b/tests/unit/account-rotation.test.ts @@ -7,6 +7,8 @@ import { markSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, type RotatableAccount, } from "../../open-sse/executors/accountRotation.ts"; @@ -114,3 +116,100 @@ describe("accountRotation", () => { assert.strictEqual(isNetworkErrorRotatable(withoutProxy), false); }); }); + +describe("isEmptyUpstreamRejection", () => { + it("matches the observed malformed completion envelope (no error field, empty content, null finish_reason)", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(isEmptyUpstreamRejection(400, observed), true); + }); + + it("does not match a non-400 status", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(isEmptyUpstreamRejection(200, observed), false); + assert.strictEqual(isEmptyUpstreamRejection(429, observed), false); + assert.strictEqual(isEmptyUpstreamRejection(502, observed), false); + }); + + it("does not match when an error field is present", () => { + const withError = JSON.stringify({ + error: { message: "bad request", type: "invalid_request_error" }, + }); + assert.strictEqual(isEmptyUpstreamRejection(400, withError), false); + const emptyError = JSON.stringify({ error: {} }); + assert.strictEqual(isEmptyUpstreamRejection(400, emptyError), false); + }); + + it("does not match when content is non-empty or tool_calls present", () => { + const nonEmpty = JSON.stringify({ + choices: [{ message: { role: "assistant", content: "hi" }, finish_reason: "stop" }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, nonEmpty), false); + + const toolCalls = JSON.stringify({ + choices: [ + { message: { role: "assistant", tool_calls: [{ id: "x" }] }, finish_reason: "tool_calls" }, + ], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, toolCalls), false); + }); + + it("does not match when content is a non-string non-null value (number, block array)", () => { + const numericContent = JSON.stringify({ + choices: [{ message: { role: "assistant", content: 123 }, finish_reason: null }], + }); + assert.strictEqual( + isEmptyUpstreamRejection(400, numericContent), + false, + "non-string non-null content is not eligible" + ); + + const reasoningContent = JSON.stringify({ + choices: [ + { message: { role: "assistant", reasoning_content: "thinking" }, finish_reason: null }, + ], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, reasoningContent), false); + }); + + it("does not match when choices or message are absent", () => { + const noChoices = JSON.stringify({ id: "chatcmpl_x", model: "muse" }); + assert.strictEqual(isEmptyUpstreamRejection(400, noChoices), false); + const noMessage = JSON.stringify({ choices: [{ finish_reason: null }] }); + assert.strictEqual(isEmptyUpstreamRejection(400, noMessage), false); + }); + + it("does not match when finish_reason is a literal value (not null)", () => { + const stopReason = JSON.stringify({ + choices: [{ message: { role: "assistant" }, finish_reason: "stop" }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, stopReason), false); + }); + + it("matches an empty string content (treated as eligible)", () => { + const emptyContent = JSON.stringify({ + choices: [{ message: { role: "assistant", content: "" }, finish_reason: null }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, emptyContent), true); + }); + + it("returns false for unparseable JSON rather than throwing", () => { + assert.strictEqual(isEmptyUpstreamRejection(400, "not json"), false); + assert.strictEqual(isEmptyUpstreamRejection(400, ""), false); + }); +}); + +describe("extractChatcmplId", () => { + it("extracts the chatcmpl id from an observed envelope", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(extractChatcmplId(observed), "chatcmpl_44fn2g6e7kk"); + }); + + it("falls back to 'unknown' when no id is present", () => { + assert.strictEqual(extractChatcmplId("{choices:[]}"), "unknown"); + assert.strictEqual(extractChatcmplId(""), "unknown"); + assert.strictEqual(extractChatcmplId("not json"), "unknown"); + }); +}); diff --git a/tests/unit/authz/pipeline.test.ts b/tests/unit/authz/pipeline.test.ts index 34703f270ecf..6f6469e804eb 100644 --- a/tests/unit/authz/pipeline.test.ts +++ b/tests/unit/authz/pipeline.test.ts @@ -306,6 +306,7 @@ test("runAuthzPipeline rejects new API requests during shutdown drain", async () assert.equal(response.status, 503); assert.equal(body.error.code, "SERVICE_UNAVAILABLE"); + assert.equal(response.headers.get("retry-after"), "5"); }); test("runAuthzPipeline rejects rewritten API aliases during shutdown drain", async () => { @@ -319,6 +320,7 @@ test("runAuthzPipeline rejects rewritten API aliases during shutdown drain", asy assert.equal(response.status, 503); assert.equal(response.headers.get("x-omniroute-route-class"), "CLIENT_API"); assert.equal(body.error.code, "SERVICE_UNAVAILABLE"); + assert.equal(response.headers.get("retry-after"), "5"); }); test("runAuthzPipeline allows dashboard sessions to read model catalog aliases", async () => { diff --git a/tests/unit/build/optional-transformers-dependency.test.ts b/tests/unit/build/optional-transformers-dependency.test.ts index 7ad60fb09c4b..cf5997a5f598 100644 --- a/tests/unit/build/optional-transformers-dependency.test.ts +++ b/tests/unit/build/optional-transformers-dependency.test.ts @@ -9,40 +9,52 @@ function readJson>(relPath: string): T { return JSON.parse(readFileSync(join(repoRoot, relPath), "utf8")) as T; } -test("@huggingface/transformers is a regular dependency so npm ci never skips it", () => { - // #9962 deliberately moved @huggingface/transformers out of optionalDependencies: - // as an optional dep, npm silently skipped the whole subtree on Node 24/26 (old - // pin dragged onnxruntime-node@1.21.0 whose NAN build no longer compiles), which - // broke `npm ci`/`next build` with "Can't resolve @huggingface/transformers" - // (lazy import in src/lib/memory/embedding/transformersLocal.ts). As a regular - // dep with onnxruntime-node@~1.24.3 (napi prebuilds, no node-gyp) it stays - // installable and the memory embedding path requires() cleanly. +test("ONNX chain (@huggingface/transformers + onnxruntime-node) stays optional so Termux/Android installs succeed", () => { + // #11095: onnxruntime-node declares os ["win32","darwin","linux"], so while + // these lived in `dependencies` every npm install on Android/Termux aborted + // with a fatal EBADPLATFORM. As optionalDependencies npm skips only the + // unsupported-platform subtree (with a warning) and installs normally + // everywhere else. This deliberately reverses the MECHANISM of #9962 while + // keeping its goal: #9962's skip happened because the old onnxruntime-node@ + // 1.21.0 pin built from source (NAN) and failed to compile on Node 24/26; + // the current 1.24.3 pin ships napi prebuilds, so on supported platforms the + // chain always installs and `npm ci`/`next build` keep resolving it. On + // platforms where it IS skipped, both consumers degrade gracefully via lazy/ + // dynamic imports (asserted below). const pkg = readJson<{ dependencies?: Record; optionalDependencies?: Record; + overrides?: Record; }>("package.json"); assert.equal( pkg.dependencies?.["@huggingface/transformers"], + undefined, + "transformers must NOT be a hard dependency (fatal EBADPLATFORM on Android)" + ); + assert.equal( + pkg.optionalDependencies?.["@huggingface/transformers"], "^4.2.0", - "transformers must be a regular dependency (never optional) so npm ci cannot skip it" + "transformers must be an optionalDependency" ); - assert.equal(pkg.optionalDependencies?.["@huggingface/transformers"], undefined); -}); - -test("transformers + onnxruntime-node are regular dependencies (not optional)", () => { - const pkg = readJson<{ - dependencies?: Record; - optionalDependencies?: Record; - }>("package.json"); - assert.equal( pkg.dependencies?.["onnxruntime-node"], + undefined, + "onnxruntime-node must NOT be a hard dependency (fatal EBADPLATFORM on Android)" + ); + assert.equal( + pkg.optionalDependencies?.["onnxruntime-node"], + "1.24.3", + "onnxruntime-node must be an optionalDependency pinned in lockstep with the overrides pin" + ); + assert.equal( + pkg.overrides?.["onnxruntime-node"], "1.24.3", - "onnxruntime-node is a regular dep (napi prebuilds, installable on Node 24/26)" + "the overrides pin must stay aligned with @huggingface/transformers' own pin (single-copy invariant)" ); - assert.equal(pkg.optionalDependencies?.["onnxruntime-node"], undefined); +}); +test("lockfile marks the whole ONNX chain optional", () => { const lock = readJson<{ packages: Record< string, @@ -51,26 +63,61 @@ test("transformers + onnxruntime-node are regular dependencies (not optional)", dependencies?: Record; optionalDependencies?: Record; } - >; + >; }>("package-lock.json"); assert.equal( - lock.packages[""]?.dependencies?.["@huggingface/transformers"], + lock.packages[""]?.optionalDependencies?.["@huggingface/transformers"], "^4.2.0", - "root lock dependencies must hold transformers as a regular (non-optional) dep" + "root lock optionalDependencies must hold transformers" + ); + assert.equal( + lock.packages[""]?.optionalDependencies?.["onnxruntime-node"], + "1.24.3", + "root lock optionalDependencies must hold onnxruntime-node" ); - // Optional flag is only written `true` for genuinely optional packages; - // regular deps leave it absent/null. Assert each is NOT optional. assert.ok( - !lock.packages["node_modules/@huggingface/transformers"]?.optional, - "transformers must not be marked optional in the lockfile" + lock.packages["node_modules/@huggingface/transformers"]?.optional, + "transformers must be marked optional in the lockfile" ); assert.ok( - !lock.packages["node_modules/onnxruntime-node"]?.optional, - "onnxruntime-node must not be marked optional in the lockfile" + lock.packages["node_modules/onnxruntime-node"]?.optional, + "onnxruntime-node must be marked optional in the lockfile" ); assert.ok( - !lock.packages["node_modules/onnxruntime-common"]?.optional, - "onnxruntime-common must not be marked optional in the lockfile" + lock.packages["node_modules/onnxruntime-common"]?.optional, + "onnxruntime-common must be marked optional in the lockfile" + ); +}); + +test("every @huggingface/transformers consumer loads it lazily so absent installs degrade gracefully", () => { + // If any module ever switches to a STATIC import of the optional chain, + // startup crashes on platforms where npm skipped it (Android/Termux). + // transformersLocal.ts must keep its lazy await import() (D8/D25); + // onnxWorker.ts must keep its runtime-variable dynamicImport indirection. + + const embeddingSrc = readFileSync( + join(repoRoot, "src/lib/memory/embedding/transformersLocal.ts"), + "utf8" + ); + assert.doesNotMatch( + embeddingSrc, + /^\s*import\s+(?:[^'"]*?\s+from\s+)?["']@huggingface\/transformers["']/m, + "transformersLocal.ts must not statically import @huggingface/transformers" + ); + assert.match( + embeddingSrc, + /await import\(["']@huggingface\/transformers["']\)/, + "transformersLocal.ts must load @huggingface/transformers via await import()" + ); + + const workerSrc = readFileSync( + join(repoRoot, "open-sse/services/compression/engines/llmlingua/onnxWorker.ts"), + "utf8" + ); + assert.doesNotMatch( + workerSrc, + /^\s*import\s+(?:[^'"]*?\s+from\s+)?["']@huggingface\/transformers["']/m, + "onnxWorker.ts must not statically import @huggingface/transformers" ); }); diff --git a/tests/unit/call-logs-row-filter.test.ts b/tests/unit/call-logs-row-filter.test.ts new file mode 100644 index 000000000000..24090389f6e8 --- /dev/null +++ b/tests/unit/call-logs-row-filter.test.ts @@ -0,0 +1,47 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { rowMatchesFilter } from "../../src/app/api/usage/call-logs/route.ts"; + +test.describe("call-logs rowMatchesFilter unit tests", () => { + const baseRow = { + id: "log-1", + status: 200, + model: "openai/gpt-4o", + provider: "openai", + providerDisplay: "OpenAI Main", + account: "Work Account", + apiKeyName: "DevKey", + comboName: "SmartRouter", + correlationId: "corr-12345", + path: "/v1/chat/completions", + error: null, + }; + + test("status filter matches ok, error, and explicit status codes", () => { + assert.equal(rowMatchesFilter(baseRow, { status: "ok" }), true); + assert.equal(rowMatchesFilter(baseRow, { status: "error" }), false); + assert.equal(rowMatchesFilter(baseRow, { status: 200 }), true); + assert.equal(rowMatchesFilter(baseRow, { status: 500 }), false); + + const errorRow = { ...baseRow, status: 500, error: "Internal Error" }; + assert.equal(rowMatchesFilter(errorRow, { status: "ok" }), false); + assert.equal(rowMatchesFilter(errorRow, { status: "error" }), true); + }); + + test("provider filter matches provider name and excludes mismatched in-memory rows", () => { + assert.equal(rowMatchesFilter(baseRow, { provider: "openai" }), true); + assert.equal(rowMatchesFilter(baseRow, { provider: "anthropic" }), false); + }); + + test("model filter matches model name and excludes mismatched in-memory rows", () => { + assert.equal(rowMatchesFilter(baseRow, { model: "gpt-4o" }), true); + assert.equal(rowMatchesFilter(baseRow, { model: "claude-3-5-sonnet" }), false); + }); + + test("search query matches across haystack fields", () => { + assert.equal(rowMatchesFilter(baseRow, { search: "SmartRouter" }), true); + assert.equal(rowMatchesFilter(baseRow, { search: "DevKey" }), true); + assert.equal(rowMatchesFilter(baseRow, { search: "corr-12345" }), true); + assert.equal(rowMatchesFilter(baseRow, { search: "non-existent" }), false); + }); +}); diff --git a/tests/unit/capture-critical-db-state.test.ts b/tests/unit/capture-critical-db-state.test.ts index 0194b5c92f60..fcf2a017b7e6 100644 --- a/tests/unit/capture-critical-db-state.test.ts +++ b/tests/unit/capture-critical-db-state.test.ts @@ -6,12 +6,13 @@ import path from "node:path"; // Shared across all tests — the module caches DATA_DIR / SQLITE_FILE at load time, // so we must create the temp dir and import exactly once. +type CoreModule = typeof import("../../src/lib/db/core.ts"); let tempDir: string; let originalDataDir: string | undefined; -let getDbInstance: any; -let resetDbInstance: any; -let ensureDbInitialized: any; -let closeDbInstance: any; +let getDbInstance: CoreModule["getDbInstance"]; +let resetDbInstance: CoreModule["resetDbInstance"]; +let ensureDbInitialized: CoreModule["ensureDbInitialized"]; +let closeDbInstance: CoreModule["closeDbInstance"]; before(async () => { tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-db-test-")); diff --git a/tests/unit/chat-body-admission.test.ts b/tests/unit/chat-body-admission.test.ts index e4707402a159..05afae869c82 100644 --- a/tests/unit/chat-body-admission.test.ts +++ b/tests/unit/chat-body-admission.test.ts @@ -16,6 +16,7 @@ const { resolveSelfLoopBearer, } = admissionModule; const { withEarlyStreamKeepalive } = await import("../../open-sse/utils/earlyStreamKeepalive.ts"); +const { getActiveRequestCount } = await import("../../src/lib/gracefulShutdown.ts"); /** * Save/restore the env-var keys that `resolveSelfLoopBearer` reads so tests can @@ -49,6 +50,25 @@ function chatRequest(body: string, contentLength: string | null = String(body.le }); } +test("heavyweight leases are counted for SIGTERM drain (#11015)", () => { + globalThis.__omnirouteShutdown = { init: true, shuttingDown: false, activeRequests: 0 }; + const controller = new ChatAdmissionController(2); + const before = getActiveRequestCount(); + const lease = controller.tryAcquireHeavy(); + assert.ok(lease); + assert.equal(getActiveRequestCount(), before + 1); + const headroom = controller.tryAcquireHealthyHeadroom(); + assert.ok(headroom); + assert.equal(getActiveRequestCount(), before + 2); + lease.release(); + assert.equal(getActiveRequestCount(), before + 1); + headroom.release(); + assert.equal(getActiveRequestCount(), before); + lease.release(); + headroom.release(); + assert.equal(getActiveRequestCount(), before); +}); + test("small known body is admitted without consuming heavyweight capacity", async () => { const controller = new ChatAdmissionController(1); const result = await admitChatRequest(chatRequest("{}"), { diff --git a/tests/unit/chatcore-translation-paths.test.ts b/tests/unit/chatcore-translation-paths.test.ts index 08cc564cc4e5..6ecb0487889c 100644 --- a/tests/unit/chatcore-translation-paths.test.ts +++ b/tests/unit/chatcore-translation-paths.test.ts @@ -369,7 +369,7 @@ async function invokeChatCore({ onCredentialsRefreshed = null, onRequestSuccess = null, sessionAffinityKey = null, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", managedLease = null, cachedSettings = null, }: any = {}) { @@ -631,7 +631,7 @@ test("chatCore translates a streaming Responses upstream for a Chat client", asy assert.match(streamed, /"content":"ok"/); assert.match(streamed, /data: \[DONE\]/); }); -test("chatCore rejects opaque reasoning for unknown Responses targets unless explicitly enabled", async () => { +test("chatCore drops opaque reasoning for plaintext Responses targets by default (#10959)", async () => { const reasoningItems = [ { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }, { type: "reasoning", encrypted_content: "" }, @@ -640,7 +640,7 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp { id: "fc_call", type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, ]; - const rejected = await invokeChatCore({ + const dropped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/responses", @@ -656,9 +656,12 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp responseFormat: "openai-responses", }); - assert.equal(rejected.result.success, false); - assert.equal(rejected.result.status, 400); - assert.equal(rejected.calls.length, 0); + assert.equal(dropped.result.success, true); + assert.equal(dropped.calls.length, 1); + assert.deepEqual( + dropped.call.body.input.filter((item) => item.type === "reasoning"), + [{ type: "reasoning", summary: [{ text: "not self-contained" }] }] + ); const enabled = await invokeChatCore({ provider: "openai-compatible-sp-openai", @@ -682,7 +685,7 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp assert.deepEqual( input.filter((item) => item.type === "reasoning"), [ - { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }, + { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }, // summary defaulted by #11110 { type: "reasoning", summary: [{ text: "not self-contained" }] }, ] ); @@ -693,9 +696,9 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp assert.equal(input.find((item) => item.type === "function_call")?.id, undefined); }); -test("chatCore applies Chat reasoning compatibility before stream mode diverges", async () => { +test("chatCore drops incompatible Chat reasoning before stream mode diverges (#10959)", async () => { for (const stream of [false, true]) { - const rejected = await invokeChatCore({ + const dropped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/chat/completions", @@ -728,14 +731,14 @@ test("chatCore applies Chat reasoning compatibility before stream mode diverges" }, }); - assert.equal(rejected.result.success, false, `stream=${stream}`); - assert.equal(rejected.result.status, 400, `stream=${stream}`); - assert.equal(rejected.calls.length, 0, `stream=${stream}`); + assert.equal(dropped.result.success, true, `stream=${stream}`); + assert.equal(dropped.calls.length, 1, `stream=${stream}`); + assert.equal(dropped.call.body.messages[0].reasoning_details, undefined, `stream=${stream}`); } }); -test("chatCore can drop incompatible reasoning for an opted-in Combo attempt", async () => { - const dropped = await invokeChatCore({ +test("chatCore preserves Combo skip behavior for incompatible reasoning", async () => { + const skipped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/responses", @@ -757,15 +760,12 @@ test("chatCore can drop incompatible reasoning for an opted-in Combo attempt", a }, responseFormat: "openai-responses", isCombo: true, - reasoningTransportFallback: "drop", + reasoningTransportFallback: "skip", }); - assert.equal(dropped.result.success, true); - assert.equal(dropped.calls.length, 1); - assert.equal( - dropped.call.body.input.some((item) => item.type === "reasoning"), - false - ); + assert.equal(skipped.result.success, false); + assert.equal(skipped.result.status, 400); + assert.equal(skipped.calls.length, 0); }); test("chatCore carries Chat reasoning_content into official DeepSeek Responses input", async () => { @@ -800,6 +800,7 @@ test("chatCore carries Chat reasoning_content into official DeepSeek Responses i assert.deepEqual(call.body.input.slice(0, 3), [ { type: "reasoning", + summary: [], // defaulted on freshly-built reasoning items (#11129) content: [{ type: "reasoning_text", text: "Inspect before calling the tool" }], }, { @@ -866,7 +867,7 @@ test("chatCore replays nonstream DeepSeek Responses reasoning across a Chat tool assert.equal(second.result.success, true); assert.deepEqual( second.call.body.input.find((item) => item.type === "reasoning"), - { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }] } + { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], summary: [] } // summary defaulted by #11129 ); }); @@ -930,7 +931,7 @@ test("chatCore replays streamed DeepSeek Responses reasoning across a Chat tool assert.equal(second.result.success, true); assert.deepEqual( second.call.body.input.find((item) => item.type === "reasoning"), - { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }] } + { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], summary: [] } // summary defaulted by #11129 ); }); @@ -1132,7 +1133,7 @@ test("chatCore automatically preserves provider-generated opaque reasoning for C assert.equal(result.success, true); assert.deepEqual( call.body.input.filter((item) => item.type === "reasoning"), - [{ id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }] + [{ id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }] // summary defaulted by #11110 ); assert.equal( call.body.input.some((item) => item.type === "item_reference"), diff --git a/tests/unit/claude-code-tool-casing-identity-echo.test.ts b/tests/unit/claude-code-tool-casing-identity-echo.test.ts new file mode 100644 index 000000000000..63d1edf3d248 --- /dev/null +++ b/tests/unit/claude-code-tool-casing-identity-echo.test.ts @@ -0,0 +1,205 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { restoreClaudeToolName } from "../../open-sse/services/claudeCodeToolRemapper.ts"; +import { openaiToClaudeResponse } from "../../open-sse/translator/response/openai-to-claude.ts"; +import { translateNonStreamingResponse } from "../../open-sse/handlers/responseTranslator.ts"; +import { FORMATS } from "../../open-sse/translator/formats.ts"; + +interface ClaudeEvent { + type: string; + index?: number; + content_block?: { type: string; id?: string; name?: string; input?: unknown }; +} + +type TranslatorState = Record; + +function firstToolUse(events: ClaudeEvent[] | null): ClaudeEvent["content_block"] { + return events?.find( + (e) => e.type === "content_block_start" && e.content_block?.type === "tool_use" + )?.content_block; +} + +function openaiToolCallChunk(name: string): { choices: Array> } { + return { + choices: [ + { + delta: { + tool_calls: [{ index: 0, id: "call_echo", function: { name, arguments: "" } }], + }, + }, + ], + }; +} + +/** + * Claude Code 2.1.x added CronCreate/CronList/CronDelete/ScheduleWakeup/ + * EnterWorktree. The `/loop` skill schedules via CronCreate; upstream gateways + * that emit the lowercased name (and echo it into the toolNameMap alias + * channel) previously let `croncreate` reach Claude Code un-restored, which + * the CLI rejects with "No such tool available" — killing /loop AND every + * other native tool call emitted in lowercase form. + */ +describe("Claude Code cron-era tool names survive identity-echo alias maps", () => { + const ECHO_MAPS = [ + ["identity entry croncreate→croncreate", new Map([["croncreate", "croncreate"]])], + ["cloak-direction entry CronCreate→croncreate", new Map([["CronCreate", "croncreate"]])], + [ + "identity + unrelated aliases", + new Map([ + ["subdispatch", "SubDispatch"], + ["croncreate", "croncreate"], + ]), + ], + ] as const; + + for (const [label, map] of ECHO_MAPS) { + it(`restoreClaudeToolName upgrades echoed lowercase cron tools — ${label}`, () => { + assert.equal(restoreClaudeToolName("croncreate", map), "CronCreate"); + assert.equal(restoreClaudeToolName("cronlist", map), "CronList"); + assert.equal(restoreClaudeToolName("crondelete", map), "CronDelete"); + assert.equal(restoreClaudeToolName("schedulewakeup", map), "ScheduleWakeup"); + assert.equal(restoreClaudeToolName("enterworktree", map), "EnterWorktree"); + assert.equal(restoreClaudeToolName("bash", map), "Bash"); + assert.equal(restoreClaudeToolName("taskcreate", map), "TaskCreate"); + assert.equal(restoreClaudeToolName("taskupdate", map), "TaskUpdate"); + assert.equal(restoreClaudeToolName("tasklist", map), "TaskList"); + assert.equal(restoreClaudeToolName("taskget", map), "TaskGet"); + }); + + it(`openaiToClaudeResponse emits PascalCase content_block.name — ${label}`, () => { + const state: TranslatorState = { + toolCalls: new Map(), + nextBlockIndex: 0, + toolNameMap: map, + }; + const block = firstToolUse( + openaiToClaudeResponse(openaiToolCallChunk("croncreate"), state) as ClaudeEvent[] + ); + assert.equal(block?.name, "CronCreate"); + }); + } + + it("request-side non-identity alias still beats canonical casing", () => { + // A client that actually declared a custom lowercase MCP-style name keeps it. + const map = new Map([ + ["read", "mcp__fs__read"], + ["croncreate", "CronCreate"], + ]); + assert.equal(restoreClaudeToolName("read", map), "mcp__fs__read"); + }); + + it("unknown tools with identity entries are preserved verbatim", () => { + const map = new Map([["my_custom_tool", "my_custom_tool"]]); + assert.equal(restoreClaudeToolName("my_custom_tool", map), "my_custom_tool"); + }); + + it("canonical echo stays canonical on no-map routes (live repro 2026-08-22)", () => { + // Live-tested against glm-5.2 via opencode-go: the request declared + // CronCreate/Bash, the gateway echoed them TitleCase, and claude-to-openai + // builds no _toolNameMap — the old #7926 REVERSE_MAP fallback downcased + // the echo to `croncreate`/`bash`, which Claude Code rejects with + // "No such tool available", killing those tools for the whole session. + // Every restoreClaudeToolName caller converts toward a Claude-format + // client, so blind TitleCase→lowercase downcasing has no legitimate + // consumer left: legacy lowercase clients are protected by explicit + // alias maps (see test above), not by unmapped downcasing. + assert.equal(restoreClaudeToolName("TodoWrite", null), "TodoWrite"); + assert.equal(restoreClaudeToolName("Read", undefined), "Read"); + assert.equal(restoreClaudeToolName("WebSearch", null), "WebSearch"); + }); +}); + +/** + * Live-reproduced 2026-08-22: an `ox-alpha-free` upstream answered the + * stream:true /v1/messages request with a non-streaming JSON body; omniroute + * converted it via translateNonStreamingResponse, which emitted tool_use.name + * verbatim ("bash") — Claude Code rejected it with "No such tool available", + * killing Bash/Read/Write/CronCreate for the whole session. + */ +describe("translateNonStreamingResponse restores Claude Code tool casing", () => { + function openaiJson(name: string) { + return { + id: "202608221120471ad3e3bd71e24afd", + object: "chat.completion", + choices: [ + { + index: 0, + finish_reason: "tool_calls", + message: { + role: "assistant", + content: null, + reasoning_content: "The user wants echo ok", + tool_calls: [ + { + id: "call_b90e72ccc16f440c88c4f9e6", + type: "function", + function: { name, arguments: '{"command":"echo ok"}' }, + }, + ], + }, + }, + ], + usage: { prompt_tokens: 246, completion_tokens: 35 }, + }; + } + + it("upgrades lowercase native tool names with no alias map (live repro)", () => { + const out = translateNonStreamingResponse( + openaiJson("bash"), + FORMATS.OPENAI, + FORMATS.CLAUDE, + null + ); + const toolUse = out.content.find((b) => b.type === "tool_use"); + assert.equal(toolUse.name, "Bash"); + assert.equal(toolUse.input.command, "echo ok"); + }); + + it("upgrades cron-era tools through identity-echo maps", () => { + const out = translateNonStreamingResponse( + openaiJson("croncreate"), + FORMATS.OPENAI, + FORMATS.CLAUDE, + new Map([["croncreate", "croncreate"]]) + ); + assert.equal(out.content.find((b) => b.type === "tool_use").name, "CronCreate"); + }); + + it("request-side aliases still win over canonical casing", () => { + const out = translateNonStreamingResponse( + openaiJson("read"), + FORMATS.OPENAI, + FORMATS.CLAUDE, + new Map([["read", "mcp__fs__read"]]) + ); + assert.equal(out.content.find((b) => b.type === "tool_use").name, "mcp__fs__read"); + }); + + it("keeps canonical casing the upstream echoed verbatim when no alias map exists (live repro #11085)", () => { + // Live-tested on the Claude Code → OpenAI-style upstream route: the request + // declares CronCreate/Bash, the gateway echoes them TitleCase, and + // claude-to-openai builds no _toolNameMap — the #7926 REVERSE_MAP fallback + // must not downcase a canonical name back into "No such tool available". + const out = translateNonStreamingResponse( + openaiJson("CronCreate"), + FORMATS.OPENAI, + FORMATS.CLAUDE, + null + ); + assert.equal(out.content.find((b) => b.type === "tool_use").name, "CronCreate"); + assert.equal(restoreClaudeToolName("Bash", null), "Bash"); + assert.equal(restoreClaudeToolName("WebSearch", null), "WebSearch"); + assert.equal(restoreClaudeToolName("TaskCreate", new Map()), "TaskCreate"); + }); + + it("declared lowercase form still wins when an explicit alias maps canonical → lowercase", () => { + // Legacy OpenCode/XML-style clients declare `bash`; request-side cloak + // records { CronCreate→croncreate }-style aliases and restoreClaudeToolName + // must keep honoring them even when the upstream echoes the canonical form. + assert.equal( + restoreClaudeToolName("CronCreate", new Map([["CronCreate", "croncreate"]])), + "croncreate" + ); + assert.equal(restoreClaudeToolName("Read", new Map([["Read", "read"]])), "read"); + }); +}); diff --git a/tests/unit/claude-tool-name-casing-fix.test.ts b/tests/unit/claude-tool-name-casing-fix.test.ts index c02fd09286e1..5c11d1652bce 100644 --- a/tests/unit/claude-tool-name-casing-fix.test.ts +++ b/tests/unit/claude-tool-name-casing-fix.test.ts @@ -50,10 +50,13 @@ describe("Claude Code Tool Name Casing Fixes", () => { assert.equal(restoreClaudeToolName("exitplanmode"), "ExitPlanMode"); }); - it("restoreClaudeToolName keeps the #7926 TitleCase→lowercase fallback with no map", () => { - // Clients with no request-side map (XML / OpenCode-style) expect lowercase. - assert.equal(restoreClaudeToolName("TodoWrite"), "todowrite"); - assert.equal(restoreClaudeToolName("Read"), "read"); + it("restoreClaudeToolName keeps canonical TitleCase with no map (#11085 live repro)", () => { + // Live-tested 2026-08-22: no-map routes (Claude Code → OpenAI-style + // upstreams) receive the gateway's TitleCase echo and must keep it — + // downcasing made Claude Code reject its own tools. XML/OpenCode-style + // lowercase clients are protected by explicit alias maps instead. + assert.equal(restoreClaudeToolName("TodoWrite"), "TodoWrite"); + assert.equal(restoreClaudeToolName("Read"), "Read"); }); it("restoreClaudeToolName prefers toolNameMap over the static map", () => { @@ -150,9 +153,7 @@ describe("Claude Code Tool Name Casing Fixes", () => { choices: [ { delta: { - tool_calls: [ - { index: 0, id: "call_123", function: { name: "bash", arguments: "" } }, - ], + tool_calls: [{ index: 0, id: "call_123", function: { name: "bash", arguments: "" } }], }, }, ], diff --git a/tests/unit/cli-combo-create-models-10954.test.ts b/tests/unit/cli-combo-create-models-10954.test.ts index dc53459d6ed5..52fabf8c1690 100644 --- a/tests/unit/cli-combo-create-models-10954.test.ts +++ b/tests/unit/cli-combo-create-models-10954.test.ts @@ -90,7 +90,15 @@ test("combo create — parses --models without throwing (Commander option regist }); await prog.parseAsync( - ["node", "x", "combo", "create", "my-combo", "--models", "openai/gpt-4o,anthropic/claude-3-opus"], + [ + "node", + "x", + "combo", + "create", + "my-combo", + "--models", + "openai/gpt-4o,anthropic/claude-3-opus", + ], { from: "node" } ); @@ -222,3 +230,26 @@ test("combo create (HTTP) — POST /api/combos body carries the parsed models", else process.env.DATA_DIR = ORIGINAL_DATA_DIR; } }); + +// Regression for the follow-up of #11011: with --models available, creating +// an empty combo is no longer a legitimate path on either transport. +test("combo create without any model is refused before reaching a transport", async () => { + await withComboEnv(async () => { + const errors: string[] = []; + const originalError = console.error; + console.error = (msg?: unknown) => { + errors.push(String(msg)); + }; + try { + const mod = await import("../../bin/cli/commands/combo.mjs"); + const rc = await mod.runComboCreateCommand("guard-test"); + assert.equal(rc, 1); + } finally { + console.error = originalError; + } + assert.ok( + errors.some((m) => m.includes("--models")), + `stderr should name --models, got: ${errors.join(" | ")}` + ); + }); +}); diff --git a/tests/unit/cline-model-format-11099.test.ts b/tests/unit/cline-model-format-11099.test.ts new file mode 100644 index 000000000000..3e5a87de43c7 --- /dev/null +++ b/tests/unit/cline-model-format-11099.test.ts @@ -0,0 +1,39 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { getModelsByProviderId } from "../../open-sse/config/providerModels.ts"; +import { parseClineRecommendedModels } from "../../open-sse/services/clinepassModels.ts"; + +test("#11099: Cline provider catalog model IDs use valid modelType/model format", () => { + const models = getModelsByProviderId("cline"); + assert.ok(models.length > 0, "cline provider must expose models"); + + for (const model of models) { + assert.match( + model.id, + /^[a-z0-9-]+-?[a-z0-9-]*\/[a-z0-9._:-]+$/i, + `Model ID '${model.id}' must follow provider/model format` + ); + assert.notEqual( + model.id.split("/")[0], + "zai", + "Model ID must use 'z-ai' instead of invalid 'zai'" + ); + } +}); + +test("#11099: parseClineRecommendedModels correctly extracts recommended/free models", () => { + const mockPayload = { + recommended: [ + { id: "moonshotai/kimi-k3", name: "kimi-k3" }, + { id: "x-ai/grok-4.5", name: "grok-4.5" }, + ], + free: [{ id: "deepseek/deepseek-v4-flash", name: "deepseek-v4-flash" }], + }; + + const parsed = parseClineRecommendedModels(mockPayload); + assert.equal(parsed.length, 3); + assert.equal(parsed[0].id, "moonshotai/kimi-k3"); + assert.equal(parsed[1].id, "x-ai/grok-4.5"); + assert.equal(parsed[2].id, "deepseek/deepseek-v4-flash"); +}); diff --git a/tests/unit/codex-drop-nonstandard-events.test.ts b/tests/unit/codex-drop-nonstandard-events.test.ts index d2aff5481498..dec24970ca5a 100644 --- a/tests/unit/codex-drop-nonstandard-events.test.ts +++ b/tests/unit/codex-drop-nonstandard-events.test.ts @@ -17,6 +17,19 @@ function sseResponse(body: string): Response { }); } +function chunkedSseResponse(chunks: string[]): Response { + const encoder = new TextEncoder(); + return new Response( + new ReadableStream({ + start(controller) { + for (const chunk of chunks) controller.enqueue(encoder.encode(chunk)); + controller.close(); + }, + }), + { status: 200, headers: { "content-type": "text/event-stream" } } + ); +} + async function readAll(res: Response): Promise { return await res.text(); } @@ -61,10 +74,10 @@ describe("codexDropNonstandardEvents (#11014)", () => { describe("filterNonstandardCodexSse (#4715)", () => { it("drops codex.* event blocks but keeps standard response.* events", async () => { const stream = - "event: response.created\ndata: {\"type\":\"response.created\"}\n\n" + + 'event: response.created\ndata: {"type":"response.created"}\n\n' + "event: codex.rate_limits\n\n" + - "event: response.output_text.delta\ndata: {\"delta\":\"hi\"}\n\n" + - "event: response.completed\ndata: {\"type\":\"response.completed\"}\n\n"; + 'event: response.output_text.delta\ndata: {"delta":"hi"}\n\n' + + 'event: response.completed\ndata: {"type":"response.completed"}\n\n'; const out = await readAll(filterNonstandardCodexSse(sseResponse(stream))); assert.ok(!out.includes("codex.rate_limits"), "codex.* frame must be stripped"); assert.ok(out.includes("response.created"), "standard events preserved"); @@ -72,18 +85,31 @@ describe("filterNonstandardCodexSse (#4715)", () => { assert.ok(out.includes("response.completed"), "terminal event preserved"); }); + it("filters CRLF-framed events split across transport chunks", async () => { + const response = chunkedSseResponse([ + 'event: response.created\r\ndata: {"type":"response.created"}\r\n\r', + "\nevent: codex.rate_limits\r\n\r\n", + 'event: response.completed\r\ndata: {"type":"response.completed"}\r\n\r\n', + ]); + + const out = await readAll(filterNonstandardCodexSse(response)); + + assert.ok(!out.includes("codex.rate_limits"), "codex.* frame must be stripped"); + assert.ok(out.includes("response.created"), "standard events preserved"); + assert.ok(out.includes("response.completed"), "terminal event preserved"); + }); + it("passes through non-SSE responses untouched", async () => { - const json = new Response("{\"ok\":true}", { + const json = new Response('{"ok":true}', { status: 200, headers: { "content-type": "application/json" }, }); const out = filterNonstandardCodexSse(json); - assert.equal(await out.text(), "{\"ok\":true}"); + assert.equal(await out.text(), '{"ok":true}'); }); it("drops a trailing codex.* block with no double-newline terminator (flush path)", async () => { - const stream = - "event: response.created\ndata: {}\n\n" + "event: codex.token_count\ndata: {}"; + const stream = "event: response.created\ndata: {}\n\n" + "event: codex.token_count\ndata: {}"; const out = await readAll(filterNonstandardCodexSse(sseResponse(stream))); assert.ok(out.includes("response.created")); assert.ok(!out.includes("codex.token_count")); diff --git a/tests/unit/columns-validation.test.ts b/tests/unit/columns-validation.test.ts new file mode 100644 index 000000000000..41f70cddeb25 --- /dev/null +++ b/tests/unit/columns-validation.test.ts @@ -0,0 +1,24 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + sanitizeRateLimitOverrides, + sanitizeQuotaWindowThresholds, +} from "@/lib/db/providers/columns"; + +test("sanitizeRateLimitOverrides surfaces rejected keys (blocking, not silent)", () => { + const r = sanitizeRateLimitOverrides({ rpm: 10, foo: 1, tpm: -1 }); + assert.deepEqual(r.sanitized, { rpm: 10 }); + assert.deepEqual(r.rejected.sort(), ["foo", "tpm"]); +}); + +test("sanitizeQuotaWindowThresholds surfaces key-too-long and out-of-range", () => { + const r = sanitizeQuotaWindowThresholds({ ["a".repeat(65)]: 50, win: 101 }); + assert.ok(r.rejected.length >= 1); + assert.ok(r.rejected.includes("win")); +}); + +test("valid input yields no rejected keys", () => { + const r = sanitizeRateLimitOverrides({ rpm: 10, tpm: 20 }); + assert.deepEqual(r.rejected, []); + assert.deepEqual(r.sanitized, { rpm: 10, tpm: 20 }); +}); diff --git a/tests/unit/combo-bracket-names.test.ts b/tests/unit/combo-bracket-names.test.ts index 32844173cc15..0f4d8af0de92 100644 --- a/tests/unit/combo-bracket-names.test.ts +++ b/tests/unit/combo-bracket-names.test.ts @@ -30,6 +30,7 @@ test.after(() => { test("combo schemas accept names with spaces and square brackets", () => { const createResult = schemas.createComboSchema.safeParse({ name: "Claude [1m]", + models: ["anthropic/claude-3-opus"], }); const updateResult = schemas.updateComboSchema.safeParse({ name: "Claude [1m]", diff --git a/tests/unit/combo-context-length.test.ts b/tests/unit/combo-context-length.test.ts index f47539d64ad2..c416b4d8be38 100644 --- a/tests/unit/combo-context-length.test.ts +++ b/tests/unit/combo-context-length.test.ts @@ -46,6 +46,7 @@ test.after(async () => { test("createComboSchema accepts valid context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 128000, }); assert.equal(result.success, true); @@ -54,6 +55,7 @@ test("createComboSchema accepts valid context_length", () => { test("createComboSchema rejects context_length below minimum (1000)", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 999, }); assert.equal(result.success, false); @@ -62,6 +64,7 @@ test("createComboSchema rejects context_length below minimum (1000)", () => { test("createComboSchema rejects context_length above maximum (2000000)", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 2000001, }); assert.equal(result.success, false); @@ -70,12 +73,14 @@ test("createComboSchema rejects context_length above maximum (2000000)", () => { test("createComboSchema accepts context_length at exact boundaries", () => { const min = schemas.createComboSchema.safeParse({ name: "MinCombo", + models: ["openai/gpt-4o-mini"], context_length: 1000, }); assert.equal(min.success, true); const max = schemas.createComboSchema.safeParse({ name: "MaxCombo", + models: ["openai/gpt-4o-mini"], context_length: 2000000, }); assert.equal(max.success, true); @@ -84,6 +89,7 @@ test("createComboSchema accepts context_length at exact boundaries", () => { test("createComboSchema rejects non-integer context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 128000.5, }); assert.equal(result.success, false); @@ -92,6 +98,7 @@ test("createComboSchema rejects non-integer context_length", () => { test("createComboSchema accepts omitted context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], }); assert.equal(result.success, true); }); diff --git a/tests/unit/combo-empty-models.test.ts b/tests/unit/combo-empty-models.test.ts index 1486628ec537..95f32813a34c 100644 --- a/tests/unit/combo-empty-models.test.ts +++ b/tests/unit/combo-empty-models.test.ts @@ -24,9 +24,9 @@ test("an update cannot remove every model from a combo", () => { assert.equal(updateComboSchema.safeParse({ name: "renamed" }).success, true); }); -test("creating a combo with no model stays allowed — the CLI does it on purpose", () => { - assert.equal(createComboSchema.safeParse({ name: "drafted", models: [] }).success, true); - assert.equal(createComboSchema.safeParse({ name: "drafted" }).success, true); +test("creating a combo without a model is refused at the boundary", () => { + assert.equal(createComboSchema.safeParse({ name: "drafted", models: [] }).success, false); + assert.equal(createComboSchema.safeParse({ name: "drafted" }).success, false); }); test("the copilot createCombo tool stores targets where the router looks for them", async () => { diff --git a/tests/unit/combo-health-autopilot-counter.test.ts b/tests/unit/combo-health-autopilot-counter.test.ts new file mode 100644 index 000000000000..1bd8330753cd --- /dev/null +++ b/tests/unit/combo-health-autopilot-counter.test.ts @@ -0,0 +1,108 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; + +import type { + ComboForecastResponse, + ComboHealthResponse, + ProviderAutopilotReport, +} from "../../src/shared/types/utilization.ts"; +import { buildComboHealthAutopilotReport } from "../../src/lib/monitoring/comboHealthAutopilot.ts"; + +function healthResponse(): ComboHealthResponse { + return { + timeRange: "24h", + combos: [ + { + comboId: "c1", + comboName: "my-combo", + strategy: "fallback", + models: [], + targetHealth: [ + { + executionKey: "e1", + stepId: "s1", + model: "m", + provider: "p", + connectionId: null, + label: null, + requests: 5, + successRate: 90, + avgLatencyMs: 100, + lastStatus: "error", + lastUsedAt: null, + quotaRemainingPct: 50, + quotaIsExhausted: false, + quotaTrend: "stable", + quotaScope: "provider", + }, + ], + quotaHealth: { providers: [], worstRemainingPct: 100 }, + usageSkew: { modelDistribution: [], giniCoefficient: 0 }, + performance: { avgLatencyMs: 100, successRate: 1.0, totalRequests: 10 }, + }, + ], + }; +} + +function forecastResponse(): ComboForecastResponse { + return { + timeRange: "24h", + horizon: "30d", + asOf: new Date(0).toISOString(), + method: "linear_history", + combos: [], + }; +} + +function providerHealthResponse(): ProviderAutopilotReport { + return { providers: [] } as unknown as ProviderAutopilotReport; +} + +function buildOptions() { + return { + range: "24h" as const, + horizon: "30d" as const, + healthResponse: healthResponse(), + forecastResponse: forecastResponse(), + providerHealthResponse: providerHealthResponse(), + }; +} + +describe("combo health autopilot counter", () => { + it("exposes suggestionCount and keeps actionableCount alias", async () => { + const report = await buildComboHealthAutopilotReport(buildOptions()); + assert.equal(typeof report.summary.suggestionCount, "number"); + assert.equal(report.summary.actionableCount, report.summary.suggestionCount); + const expected = report.combos.reduce( + (sum, combo) => + sum + combo.issues.reduce((issueSum, issue) => issueSum + issue.actions.length, 0), + 0 + ); + assert.equal(report.summary.suggestionCount, expected); + }); + + it("run_combo_test action links the dashboard with the combo id", async () => { + const report = await buildComboHealthAutopilotReport(buildOptions()); + const actions = report.combos.flatMap((combo) => combo.issues.flatMap((i) => i.actions)); + const runTest = actions.find((a) => a.type === "run_combo_test"); + assert.ok(runTest, "run_combo_test action should exist"); + assert.equal(typeof runTest.href, "string"); + assert.ok(runTest.href?.includes("c1"), "href must carry the combo id"); + assert.equal( + runTest.href?.includes("/api/combos/test?comboId="), + false, + "href must not target the GET-only API route (405)" + ); + }); + + it("keeps every action in manual mode", async () => { + const report = await buildComboHealthAutopilotReport(buildOptions()); + for (const combo of report.combos) { + for (const issue of combo.issues) { + for (const action of issue.actions) { + assert.equal(action.mode, "manual"); + } + } + } + }); +}); diff --git a/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts b/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts index 0f4abd534e64..a646744871c4 100644 --- a/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts +++ b/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts @@ -15,6 +15,7 @@ import { describe, it, before } from "node:test"; import assert from "node:assert/strict"; import { ccrEngine, + getCcrStoreStats, resetCcrStore, retrieveBlock, } from "../../../open-sse/services/compression/engines/ccr/index.ts"; @@ -54,39 +55,117 @@ describe("issue #7746 — CCR must not reduce the sole user prompt to a bare, un }); it("prompt fixture is realistically sized (>= default 600-char minChars)", () => { - assert.ok(REPORTER_PROMPT.length >= 600, `fixture must be >= 600 chars, got ${REPORTER_PROMPT.length}`); + assert.ok( + REPORTER_PROMPT.length >= 600, + `fixture must be >= 600 chars, got ${REPORTER_PROMPT.length}` + ); }); - it("does not leave the model with only the bare CCR marker when no retrieve tool is available", () => { + it("non-MCP caller: CCR skips entirely — the sole user prompt passes through verbatim", () => { resetCcrStore(); const body = makeOpenCodeStyleRequestBody(); const result = ccrEngine.apply(body as Record, { stepConfig: {} }); - assert.equal(result.compressed, true, "CCR compressed the sole user message (reproducing the report)"); - - const messages = result.body.messages as Array<{ role: string; content: string }>; - const compressedContent = messages[0].content; - const isBareMarkerOnly = /^\[CCR retrieve hash=[0-9a-f]{24} chars=\d+\]$/.test(compressedContent); - + // #7746 follow-up (forge review outage, 2026-08-22): the preamble guard was + // not enough — a non-MCP caller received "[CCR retrieve hash=...] markers" + // it had no tool to resolve (upstream saw 112 of ~3.6K tokens). The engine + // now refuses to replace content at all when tools[] lacks + // omniroute_ccr_retrieve: compressed=false, message content untouched. assert.equal( - isBareMarkerOnly, + result.compressed, false, - "BUG #7746: CCR replaced the ENTIRE sole user message with nothing but the bare " + - `[CCR retrieve hash=...] marker, permanently losing the original prompt for any ` + - `non-MCP caller that cannot resolve the marker. Got: ${JSON.stringify(compressedContent)}` + "CCR must not compress for a caller without the retrieve tool" ); + assert.equal(result.stats, null, "no stats when the engine is skipped"); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].role, "user", "message role must stay user"); + assert.equal( + messages[0].content, + REPORTER_PROMPT, + "sole user prompt must pass through verbatim" + ); + assert.equal(messages.length, 1, "no protocol instruction may be injected for non-MCP callers"); + // Guard regression check: if callerSupportsCcrRetrieve ever returned true + // here, the store would silently accumulate blocks no non-MCP caller can + // retrieve. After a skip the store must hold nothing for this principal. + assert.equal(getCcrStoreStats().entries, 0, "store must stay empty after a non-MCP skip"); }); - it("the original prompt remains fully retrievable by hash even after the guard applies", () => { + // tools:[] and unrelated tools are distinct caller shapes that must all be + // treated as non-MCP: an empty array and a foreign tool list both mean the + // retrieve tool is unreachable. + for (const label of ["empty tools array", "unrelated tools"] as const) { + it(`non-MCP caller with ${label}: CCR skips entirely`, () => { + resetCcrStore(); + const tools = + label === "empty tools array" + ? [] + : [ + { type: "function", function: { name: "get_weather" } }, + { type: "function", function: { name: "web_search" } }, + ]; + const body = { ...makeOpenCodeStyleRequestBody(), tools }; + const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + + assert.equal(result.compressed, false, `${label} must not compress`); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].content, REPORTER_PROMPT, "prompt passes through verbatim"); + assert.equal(messages.length, 1, "no protocol instruction injected"); + }); + } + + // A malformed body (tools as a non-array, or entries of unexpected shape) + // must fail OPEN — no compression, never a throw into the request pipeline. + for (const malformed of [ + { tools: "not-an-array" }, + { tools: [null, 42, "x"] }, + { tools: [{}, { type: "function" }] }, + ]) { + it(`malformed tools payload (${JSON.stringify(malformed.tools)}): engine skips without throwing`, () => { + resetCcrStore(); + const body = { ...makeOpenCodeStyleRequestBody(), ...malformed }; + const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + + assert.equal(result.compressed, false, "malformed tools must fail open (skip)"); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].content, REPORTER_PROMPT, "prompt passes through verbatim"); + assert.equal( + getCcrStoreStats().entries, + 0, + "store must stay empty after a malformed-tools skip" + ); + }); + } + + it("MCP-capable caller (tools[] advertises omniroute_ccr_retrieve): replacement still runs and stays retrievable", () => { resetCcrStore(); - const body = makeOpenCodeStyleRequestBody(); + const body = { + ...makeOpenCodeStyleRequestBody(), + tools: [{ type: "function", function: { name: "omniroute_ccr_retrieve" } }], + }; const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + assert.equal(result.compressed, true, "CCR still compresses for MCP-capable callers"); const messages = result.body.messages as Array<{ role: string; content: string }>; - const compressedContent = messages[0].content; + // The protocol instruction is injected as a leading system message, so the + // compressed conversation is exactly: [instruction, original user message]. + assert.equal(messages.length, 2, "instruction + user message"); + assert.equal(messages[0].role, "system", "instruction is a leading system message"); + assert.ok( + typeof messages[0].content === "string" && messages[0].content.length > 0, + "instruction content must be non-empty" + ); + assert.ok( + messages[0].content.includes("omniroute_ccr_retrieve"), + "instruction must teach the retrieve tool contract" + ); + const compressedContent = messages[1].content; const match = compressedContent.match(/\[CCR retrieve hash=([0-9a-f]{24}) chars=\d+\]/); - assert.ok(match, "compressed content must still contain a resolvable CCR marker"); - const hash = match![1]; - assert.equal(retrieveBlock(hash), REPORTER_PROMPT, "original prompt must be stored verbatim and retrievable"); + assert.ok(match, "compressed content must contain a resolvable CCR marker"); + assert.equal( + retrieveBlock(match![1]), + REPORTER_PROMPT, + "original prompt must be stored verbatim and retrievable" + ); }); }); diff --git a/tests/unit/compression/ccr-retrieval-ramp.test.ts b/tests/unit/compression/ccr-retrieval-ramp.test.ts index 2b475e69bfb0..6e87f7b78773 100644 --- a/tests/unit/compression/ccr-retrieval-ramp.test.ts +++ b/tests/unit/compression/ccr-retrieval-ramp.test.ts @@ -78,7 +78,13 @@ describe("ccrEngine.apply — retrieval-aware compression (H8)", () => { const block = (len: number) => "x".repeat(len); const run = (content: string, retrievalRampFactor = 2) => ccrEngine.apply( - { messages: [{ role: "user", content }] }, + // The retrieve tool is advertised — this suite exercises the compression + // path itself (H8 ramp); without the tool declaration the #7746 guard + // skips the engine entirely. + { + messages: [{ role: "user", content }], + tools: [{ type: "function", function: { name: "omniroute_ccr_retrieve" } }], + }, { stepConfig: { minChars: BASE, retrievalRampFactor }, principalId: P } ); diff --git a/tests/unit/config-audit-persistence.test.ts b/tests/unit/config-audit-persistence.test.ts new file mode 100644 index 000000000000..4bad88bc5db4 --- /dev/null +++ b/tests/unit/config-audit-persistence.test.ts @@ -0,0 +1,124 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-config-audit-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; + +const core = await import("../../src/lib/db/core.ts"); +const cleanup = await import("../../src/lib/db/cleanup.ts"); +const audit = await import("../../src/domain/configAudit.ts"); + +type CountRow = { c: number }; + +function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +function countRows(): number { + const db = core.getDbInstance(); + const row = db.prepare("SELECT COUNT(*) AS c FROM config_audit_log").get() as CountRow; + return row.c; +} + +function insertOldRow(id: string, daysAgo: number) { + const db = core.getDbInstance(); + const old = new Date(Date.now() - daysAgo * 24 * 60 * 60 * 1000).toISOString(); + db.prepare( + `INSERT INTO config_audit_log + (id, timestamp, action, target, target_id, target_name, before_json, after_json, diff_json, source, note) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + ).run( + id, + old, + "update", + "provider", + "p1", + "P1", + null, + null, + JSON.stringify({ added: [], removed: [], changed: [], isEmpty: true }), + "api", + null + ); +} + +test.beforeEach(() => { + resetStorage(); +}); + +test.after(() => { + resetStorage(); +}); + +test("recordChange persists to SQLite, not memory", () => { + const db = core.getDbInstance(); + const tableRow = db + .prepare("SELECT count(*) as c FROM sqlite_master WHERE type='table' AND name='config_audit_log'") + .get() as CountRow; + assert.equal(tableRow.c, 1); + + const e = audit.recordChange("update", "provider", "p1", "My Provider", { a: 1 }, { a: 2 }, "api", null); + assert.equal(countRows(), 1); + + const { entries, total } = audit.getAuditLog({ target: "provider" }); + assert.equal(total, 1); + assert.equal(entries[0].id, e.id); + assert.deepEqual(entries[0].diff.changed, [{ key: "a", from: 1, to: 2 }]); +}); + +test("pagination + filters read from SQLite", () => { + audit.recordChange("create", "combo", "c1", "C1", null, { models: ["m1"] }, "dashboard"); + audit.recordChange("update", "combo", "c1", "C1", { models: ["m1"] }, { models: ["m1", "m2"] }, "api"); + + const { entries, total } = audit.getAuditLog({ target: "combo", limit: 1, offset: 0 }); + assert.equal(total, 2); + assert.equal(entries.length, 1); +}); + +test("getRollbackState returns the before snapshot", () => { + const e = audit.recordChange("update", "policy", "pol1", "Pol", { x: 1 }, { x: 2 }, "api"); + assert.deepEqual(audit.getRollbackState(e.id), { x: 1 }); +}); + +test("computeDiff stays pure", () => { + const d = audit.computeDiff({ a: 1 }, { a: 2, b: 3 }); + assert.deepEqual(d.added, ["b"]); + assert.deepEqual(d.changed, [{ key: "a", from: 1, to: 2 }]); +}); + +test("resetAuditLog clears persisted rows", () => { + audit.recordChange("update", "provider", "p1", "P1", { a: 1 }, { a: 2 }, "api"); + assert.equal(countRows(), 1); + audit.resetAuditLog(); + assert.equal(countRows(), 0); +}); + +test("cleanupConfigAudit prunes rows beyond retentionDays", async () => { + insertOldRow("audit-old", 40); + const r = await cleanup.cleanupConfigAudit(30); + assert.equal(r.deleted, 1); + assert.equal(countRows(), 0); +}); + +test("cleanupConfigAudit keeps recent rows within retention", async () => { + insertOldRow("audit-recent", 5); + const r = await cleanup.cleanupConfigAudit(30); + assert.equal(r.deleted, 0); + assert.equal(countRows(), 1); +}); + +test("runAutoCleanup includes a configAudit result", async () => { + insertOldRow("audit-old-2", 40); + const result = await cleanup.runAutoCleanup(); + assert.ok(result.results.configAudit); + assert.equal(typeof result.results.configAudit.deleted, "number"); + assert.equal(typeof result.results.configAudit.errors, "number"); + assert.equal(result.results.configAudit.deleted, 1); + assert.equal(countRows(), 0); +}); diff --git a/tests/unit/context-manager-purify-system-first.test.ts b/tests/unit/context-manager-purify-system-first.test.ts new file mode 100644 index 000000000000..51c658231d0f --- /dev/null +++ b/tests/unit/context-manager-purify-system-first.test.ts @@ -0,0 +1,91 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { compressContext } from "../../open-sse/services/contextManager.ts"; + +/** + * Plan-A root fix for the 2026-08-22 tokenrouter 400s. purifyHistory() used to + * splice the `[Context compressed: …]` notice as a SECOND system-role message at + * index system.length; strict gateways (TokenRouter, xiaomi-mimo/mimo) reject any + * system message at index > 0 with HTTP 400 "System message must be at the + * beginning". The notice must now merge into the leading system/developer + * message — or prepend a single system message when none exists — so the output + * never contains a system role after index 0, for ANY provider. + */ + +function bigTurn(n: number) { + return { role: "user", content: `turn ${n}: ${"x".repeat(4_000)}` }; +} + +function run(body: Record) { + // ~30k tokens of history vs a small target forces Layer-3 purify_history. + return compressContext(body, { maxTokens: 5_000, reserveTokens: 0 }); +} + +function systemIndices(messages: Array<{ role: string }>) { + return messages.map((m, i) => (m.role === "system" ? i : -1)).filter((i) => i >= 0); +} + +test("purify_history merges dropped-notice into existing leading system message", () => { + const body = { + model: "any-model", + messages: [ + { role: "system", content: "You are a helpful assistant." }, + ...Array.from({ length: 12 }, (_, i) => bigTurn(i)), + ], + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual(systemIndices(messages as Array<{ role: string }>).slice(1), []); + const first = messages[0]; + assert.equal(first.role, "system"); + const text = String(first.content); + assert.match(text, /Context compressed: \d+ earlier messages removed/); + assert.match(text, /You are a helpful assistant\./); +}); + +test("purify_history prepends a single system notice when no system message exists", () => { + const body = { + model: "any-model", + messages: Array.from({ length: 12 }, (_, i) => bigTurn(i)), + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual(systemIndices(messages as Array<{ role: string }>), [0]); + assert.match(String(messages[0].content), /Context compressed: \d+ earlier messages removed/); +}); + +test("purify_history merges into leading developer message without adding a second one", () => { + const body = { + model: "any-model", + messages: [ + { role: "developer", content: "dev instructions" }, + ...Array.from({ length: 12 }, (_, i) => bigTurn(i)), + ], + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual( + messages.filter((m) => m.role === "developer").length, + 1, + "exactly one developer message" + ); + assert.match(String(messages[0].content), /Context compressed: \d+ earlier messages removed/); + assert.match(String(messages[0].content), /dev instructions/); +}); + +test("no compression means no notice and untouched history", () => { + const body = { + model: "any-model", + messages: [ + { role: "system", content: "sys" }, + { role: "user", content: "hi" }, + ], + }; + const result = run(body); + assert.equal(result.compressed, false); + const messages = (result.body as { messages: unknown[] }).messages; + assert.equal(messages.length, 2); +}); diff --git a/tests/unit/context-manager.test.ts b/tests/unit/context-manager.test.ts index 776a3cdc256c..de95d5b6f2d3 100644 --- a/tests/unit/context-manager.test.ts +++ b/tests/unit/context-manager.test.ts @@ -75,10 +75,15 @@ for (const modelId of HYPERAGENT_FALLBACK_MODEL_IDS) { } test("getTokenLimit: does not force 1M onto non-hyperagent providers serving the same model ids", () => { - // windsurf declares an explicit per-model contextLength of 200000 for this exact id — - // a provider-unscoped substring match on "claude-opus-4" would have clobbered it to 1M. - assert.equal(getTokenLimit("windsurf", "claude-opus-4.7-max"), 200000); - // bluesminds likewise pins its own claude-opus-4-5 entry to 200000. + // windsurf used to pin this exact id at 200000, but its built-in provider entry was + // retired (#8228 — replaced by devin-desktop). With no per-provider source left, + // #11034 resolves the effort-suffixed variant via its BASE model (`claude-opus-4.7-max` + // → `claude-opus-4.7` → canonical `claude-opus-4-7`), whose real catalog window IS 1M. + // This is a legitimate base-model resolution, not the forbidden hyperagent default leak: + // it comes from the shared model catalog, never from the hyperagent registry scope. + assert.equal(getTokenLimit("windsurf", "claude-opus-4.7-max"), 1_000_000); + // bluesminds still pins its own claude-opus-4-5 entry to 200000, and that pin must win + // over both the name heuristic and any 1M window from sibling providers/catalogs. assert.equal(getTokenLimit("bluesminds", "claude-opus-4-5"), 200000); }); diff --git a/tests/unit/db-providers-split.test.ts b/tests/unit/db-providers-split.test.ts index 220544ba2d4f..284a5193c84d 100644 --- a/tests/unit/db-providers-split.test.ts +++ b/tests/unit/db-providers-split.test.ts @@ -52,26 +52,38 @@ describe("providers/columns — normalizeBooleanColumn", () => { }); describe("providers/columns — sanitizeRateLimitOverrides", () => { - it("returns null for nullish / non-object / array input", () => { - assert.equal(sanitizeRateLimitOverrides(null), null); - assert.equal(sanitizeRateLimitOverrides(undefined), null); - assert.equal(sanitizeRateLimitOverrides("x"), null); - assert.equal(sanitizeRateLimitOverrides([1, 2]), null); + it("returns {sanitized:null,rejected:[]} for nullish / non-object / array input", () => { + assert.deepEqual(sanitizeRateLimitOverrides(null), { sanitized: null, rejected: [] }); + assert.deepEqual(sanitizeRateLimitOverrides(undefined), { sanitized: null, rejected: [] }); + assert.deepEqual(sanitizeRateLimitOverrides("x"), { sanitized: null, rejected: [] }); + assert.deepEqual(sanitizeRateLimitOverrides([1, 2]), { sanitized: null, rejected: [] }); }); - it("keeps only allowed keys with non-negative integers", () => { - assert.deepEqual(sanitizeRateLimitOverrides({ rpm: 10, bogus: 5, tpm: -1 }), { rpm: 10 }); + it("keeps only allowed keys with non-negative integers, reports the rest as rejected", () => { + assert.deepEqual(sanitizeRateLimitOverrides({ rpm: 10, bogus: 5, tpm: -1 }), { + sanitized: { rpm: 10 }, + rejected: ["bogus", "tpm"], + }); }); - it("returns null when nothing valid remains", () => { - assert.equal(sanitizeRateLimitOverrides({ rpm: 1.5, nope: 3 }), null); + it("returns {sanitized:null} when nothing valid remains, with rejected keys", () => { + assert.deepEqual(sanitizeRateLimitOverrides({ rpm: 1.5, nope: 3 }), { + sanitized: null, + rejected: ["rpm", "nope"], + }); }); }); describe("providers/columns — sanitizeQuotaWindowThresholds", () => { - it("keeps only 0-100 integers", () => { - assert.deepEqual(sanitizeQuotaWindowThresholds({ a: 50, b: 120, c: 0 }), { a: 50, c: 0 }); + it("keeps only 0-100 integers, reports the rest as rejected", () => { + assert.deepEqual(sanitizeQuotaWindowThresholds({ a: 50, b: 120, c: 0 }), { + sanitized: { a: 50, c: 0 }, + rejected: ["b"], + }); }); - it("returns null when empty", () => { - assert.equal(sanitizeQuotaWindowThresholds({ a: 200 }), null); + it("returns {sanitized:null} when nothing valid remains, with rejected keys", () => { + assert.deepEqual(sanitizeQuotaWindowThresholds({ a: 200 }), { + sanitized: null, + rejected: ["a"], + }); }); }); diff --git a/tests/unit/empty-stream-no-content-8649.test.ts b/tests/unit/empty-stream-no-content-8649.test.ts index e46157fbe174..918ae7c4c065 100644 --- a/tests/unit/empty-stream-no-content-8649.test.ts +++ b/tests/unit/empty-stream-no-content-8649.test.ts @@ -252,3 +252,68 @@ test("#8649 buildStreamErrorChunks-shaped error must not be rewritten as empty c assert.match(text, /AI Model Not Found/); assert.doesNotMatch(text, /Provider returned empty content/); }); + +test("#8649 a Responses compaction-only stream is real output, not empty content", async () => { + // Codex remote compaction V2: POST /v1/responses with a compaction_trigger + // input item completes with output = [{type:"compaction", encrypted_content}] + // and no assistant text. The watcher's content keys do not include + // encrypted_content, so the healthy stream was followed by a synthetic + // response.failed ("Provider returned empty content") — which strict + // Responses clients reject even after response.completed. + const text = await runClientStream( + [ + `data: {"type":"response.in_progress"}\n\n`, + `event: response.created\ndata: ${JSON.stringify({ + type: "response.created", + response: { id: "resp_cmp", status: "in_progress", output: [] }, + })}\n\n`, + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp_cmp", + status: "completed", + output: [ + { id: "cmp_1", type: "compaction", encrypted_content: "gAAAAABencryptedpayload" }, + ], + }, + })}\n\n`, + ], + FORMATS.OPENAI_RESPONSES + ); + + assert.match(text, /"type":"compaction"/); + assert.doesNotMatch( + text, + /Provider returned empty content|response\.failed/, + "a completed compaction response must not be followed by a synthetic failure frame" + ); +}); + +test("#8649 an encrypted-reasoning-only stream is still empty content", async () => { + // Inverse of the compaction carve-out: an encrypted reasoning item is not + // user-visible output. A turn that produces only a reasoning trace and no + // message/tool call is the fake-success shape this guard exists to catch. + const text = await runClientStream( + [ + `event: response.created\ndata: ${JSON.stringify({ + type: "response.created", + response: { id: "resp_r", status: "in_progress", output: [] }, + })}\n\n`, + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp_r", + status: "completed", + output: [{ id: "rs_1", type: "reasoning", encrypted_content: "gAAAAABencryptedtrace" }], + }, + })}\n\n`, + ], + FORMATS.OPENAI_RESPONSES + ); + + assert.match( + text, + /response\.failed|Provider returned empty content/, + "a reasoning-only turn must keep tripping the empty-content guard" + ); +}); diff --git a/tests/unit/executor-kimi-web.test.ts b/tests/unit/executor-kimi-web.test.ts index 85b01957dc74..43f894e63a5f 100644 --- a/tests/unit/executor-kimi-web.test.ts +++ b/tests/unit/executor-kimi-web.test.ts @@ -1,7 +1,7 @@ -// Tests for the international Kimi web executor (www.kimi.com Connect-RPC API). +// Tests for the international Kimi web executor (www.kimi.ai Connect-RPC API). // // Previously this provider targeted kimi.moonshot.cn; that domain now redirects -// every non-CN visitor to www.kimi.com, which uses a Connect-RPC streaming API. +// every non-CN visitor to www.kimi.ai, which uses a Connect-RPC streaming API. // These tests pin the parser behavior of the Connect envelope framing and the // JSON event-delta extractor. @@ -31,7 +31,7 @@ describe("KimiWebExecutor", () => { assert.match(body.error.code, /HTTP_400|400/); }); - it("execute targets www.kimi.com (not kimi.moonshot.cn)", async () => { + it("execute targets www.kimi.ai (not kimi.moonshot.cn)", async () => { const executor = new mod.KimiWebExecutor(); let capturedUrl = ""; const originalFetch = globalThis.fetch; diff --git a/tests/unit/gemini-to-claude-tool-name-case-9008.test.ts b/tests/unit/gemini-to-claude-tool-name-case-9008.test.ts index 798e130bd6c4..99d655335ebf 100644 --- a/tests/unit/gemini-to-claude-tool-name-case-9008.test.ts +++ b/tests/unit/gemini-to-claude-tool-name-case-9008.test.ts @@ -18,8 +18,7 @@ const { claudeToGeminiRequest } = await import("../../open-sse/translator/request/claude-to-gemini.ts"); const { openaiToGeminiRequest } = await import("../../open-sse/translator/request/openai-to-gemini.ts"); -const { restoreClaudeToolName } = - await import("../../open-sse/services/claudeCodeToolRemapper.ts"); +const { restoreClaudeToolName } = await import("../../open-sse/services/claudeCodeToolRemapper.ts"); function toolUseName(events: Array> | null): string | undefined { const start = (events || []).find( @@ -27,9 +26,7 @@ function toolUseName(events: Array> | null): string | un e.type === "content_block_start" && (e.content_block as Record | undefined)?.type === "tool_use" ); - return (start?.content_block as Record | undefined)?.name as - | string - | undefined; + return (start?.content_block as Record | undefined)?.name as string | undefined; } test("#9008 restoreClaudeToolName: preserves PascalCase from the request map", () => { @@ -50,9 +47,13 @@ test("#9008 restoreClaudeToolName: maps lowercased upstream names back to declar assert.equal(restoreClaudeToolName("websearch", map), "WebSearch"); }); -test("#9008 restoreClaudeToolName: still lowercases TitleCase when no request map (#7926)", () => { - assert.equal(restoreClaudeToolName("Bash", null), "bash"); - assert.equal(restoreClaudeToolName("Read", undefined), "read"); +test("#9008 restoreClaudeToolName: keeps canonical TitleCase when no request map (#11085 live repro)", () => { + // Live-tested 2026-08-22 (glm via opencode-go → /v1/messages): the gateway + // echoed Bash/Read TitleCase and claude-to-openai ships no _toolNameMap; + // downcasing here made Claude Code reject its own tools. Legacy lowercase + // clients are protected by explicit alias maps instead of blind downcasing. + assert.equal(restoreClaudeToolName("Bash", null), "Bash"); + assert.equal(restoreClaudeToolName("Read", undefined), "Read"); }); test("#9008 Gemini → Claude: PascalCase tool_use survives when upstream echoes TitleCase", () => { diff --git a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts index d14c136e7b3a..5d927de02a8d 100644 --- a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts +++ b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts @@ -1,7 +1,7 @@ import test from "node:test"; import assert from "node:assert/strict"; -// GLM-5.3 support (released 2026-08-14, https://z.ai/blog/glm-5.3). +// GLM-5.3 support (released 2026-08-14, https://docs.z.ai/guides/llm/glm-5.3). // // Upstream ships ONE model id (`glm-5.3`) — effort is a request parameter // (`reasoning_effort`: low|high|max, default max) on the coding chat/completions @@ -12,14 +12,15 @@ import assert from "node:assert/strict"; // beta header), the 5.3 tiers use the documented `reasoning_effort` param on the // OpenAI coding transport. // -// Spec caveat: Z.ai has not yet published the default context window — 1M is -// mirrored from GLM-5.2 (same base model) per operator decision; correct when -// the official spec lands. +// Z.AI documents a 1M context window and 128K maximum output. -const { getRegistryEntry } = await import("../../open-sse/config/providerRegistry.ts"); +const { getRegistryEntry, REGISTRY } = await import("../../open-sse/config/providerRegistry.ts"); const { GlmExecutor } = await import("../../open-sse/executors/glm.ts"); const { MODEL_SPECS } = await import("../../src/shared/constants/modelSpecs.ts"); const { GLM_PRICING } = await import("../../src/shared/constants/pricing/shared-tiers.ts"); +const metadataRegistry = await import("../../src/lib/modelMetadataRegistry.ts"); +const { shouldExposeSyncedEffortVariants, SYNCED_EFFORT_SKIP_PROVIDERS } = + await import("../../open-sse/utils/syncedEffortVariants.ts"); const GLM_5_3_IDS = ["glm-5.3", "glm-5.3-high", "glm-5.3-low"] as const; @@ -38,6 +39,87 @@ function modelIds(provider: string): string[] { return (entry.models ?? []).map((m) => m.id); } +test("shared GLM providers keep their dedicated aliases instead of synthesizing another layer", () => { + for (const provider of ["glm", "glm-cn", "glmt"]) { + assert.ok(SYNCED_EFFORT_SKIP_PROVIDERS.has(provider), provider); + assert.equal( + shouldExposeSyncedEffortVariants({ + id: `${provider}/glm-5.3`, + owned_by: provider, + capabilities: { effort_tiers: ["low", "high", "max"] }, + }), + false, + provider + ); + } + assert.equal(SYNCED_EFFORT_SKIP_PROVIDERS.has("zcode"), false); +}); + +test("GLM family detection covers numeric, Z1, and bare provider model ids", () => { + for (const modelId of [ + "hf:zai-org/GLM-5.2", + "THUDM/GLM-Z1-32B-0414", + "THUDM/GLM-Z1-9B-0414", + "glm", + ]) { + assert.equal(metadataRegistry.isGlmFamilyModel(modelId), true, modelId); + } + assert.equal(metadataRegistry.isGlmFamilyModel("llama-3.3"), false); +}); + +test("catalog suppresses inferred tiers for every GLM registry entry without a provider contract", () => { + let audited = 0; + for (const [provider, entry] of Object.entries(REGISTRY)) { + for (const model of entry.models ?? []) { + if (!metadataRegistry.isGlmFamilyModel(model.id, model.name)) continue; + audited += 1; + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + if (capabilities.supportsThinking === true) { + assert.deepEqual( + capabilities.effort_tiers, + model.supportedThinkingEfforts ?? [], + `${provider}/${model.id}` + ); + } else { + assert.equal("effort_tiers" in capabilities, false, `${provider}/${model.id}`); + } + } + } + assert.ok(audited > 0); +}); + +test("catalog exposes only GLM effort tiers that each provider can route", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt", "zcode"]) { + for (const model of getRegistryEntry(provider)!.models ?? []) { + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + const expected = provider === "zcode" ? [] : (routedTiers.get(model.id) ?? []); + assert.equal(capabilities.supportsThinking, true, `${provider}/${model.id}`); + assert.deepEqual(capabilities.effort_tiers, expected, `${provider}/${model.id}`); + } + } +}); + for (const provider of ["glm", "glm-cn", "glmt"]) { test(`${provider} advertises the GLM-5.3 base model and effort tiers (GLM_SHARED_MODELS)`, () => { const ids = modelIds(provider); diff --git a/tests/unit/http-status-unprocessable-entity.test.ts b/tests/unit/http-status-unprocessable-entity.test.ts new file mode 100644 index 000000000000..636e0ebe37f1 --- /dev/null +++ b/tests/unit/http-status-unprocessable-entity.test.ts @@ -0,0 +1,7 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { HTTP_STATUS } from "../../open-sse/config/constants.ts"; + +test("HTTP_STATUS declares UNPROCESSABLE_ENTITY as 422", () => { + assert.equal(HTTP_STATUS.UNPROCESSABLE_ENTITY, 422); +}); diff --git a/tests/unit/is-local-provider-11091.test.ts b/tests/unit/is-local-provider-11091.test.ts new file mode 100644 index 000000000000..e5adc57b15a6 --- /dev/null +++ b/tests/unit/is-local-provider-11091.test.ts @@ -0,0 +1,38 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { isLocalProvider } from "../../open-sse/config/providerRegistry.ts"; + +test("isLocalProvider detects RFC1918, CGNAT/Tailscale, and mDNS private hosts", () => { + // Local / loopback + assert.equal(isLocalProvider("http://localhost:11434/v1"), true); + assert.equal(isLocalProvider("http://127.0.0.1:11434/v1"), true); + + // Docker 172.16/12 + assert.equal(isLocalProvider("http://172.18.0.2:11434/v1"), true); + + // RFC1918 LAN hosts (Issue #11091) + assert.equal(isLocalProvider("http://192.168.1.50:11434/v1"), true); + assert.equal(isLocalProvider("http://10.0.0.5:11434/v1"), true); + + // Tailscale / CGNAT (100.64/10) + assert.equal(isLocalProvider("http://100.64.1.2:11434/v1"), true); + + // Link-local (169.254/16) + assert.equal(isLocalProvider("http://169.254.1.1:11434/v1"), true); + + // mDNS / private suffixes + assert.equal(isLocalProvider("http://studio.local:11434/v1"), true); + assert.equal(isLocalProvider("http://mybox.internal:11434/v1"), true); + + // Public hosts (should be false) + assert.equal(isLocalProvider("https://api.openai.com/v1"), false); + assert.equal(isLocalProvider("https://api.anthropic.com/v1"), false); + assert.equal(isLocalProvider("http://8.8.8.8:8080/v1"), false); + + // Fails open on missing or unparseable input (Issue #11091 review finding) + assert.equal(isLocalProvider(null), false); + assert.equal(isLocalProvider(undefined), false); + assert.equal(isLocalProvider(""), false); + assert.equal(isLocalProvider("not a url"), false); + assert.equal(isLocalProvider("file:///models"), false); +}); diff --git a/tests/unit/kimi-partner-aff-links.test.ts b/tests/unit/kimi-partner-aff-links.test.ts index 2592f03fa223..2db3c56636a0 100644 --- a/tests/unit/kimi-partner-aff-links.test.ts +++ b/tests/unit/kimi-partner-aff-links.test.ts @@ -7,9 +7,8 @@ import test from "node:test"; import assert from "node:assert/strict"; const providers = await import("../../src/shared/constants/providers.ts"); -const featuredProviders = await import( - "../../src/app/(dashboard)/dashboard/providers/featuredProviders.ts" -); +const featuredProviders = + await import("../../src/app/(dashboard)/dashboard/providers/featuredProviders.ts"); const KIMI_CODING_AFF_URL = "https://www.kimi.com/code?aff=omniroute"; const KIMI_PLATFORM_AFF_URL = "https://platform.kimi.ai?aff=omniroute"; @@ -33,11 +32,11 @@ test("kimi-coding (Kimi Code CLI) top-of-page link: the Kimi Coding Plan aff lin assert.equal(kimiCoding.website, KIMI_CODING_AFF_URL); }); -test("kimi-web (Kimi Web) top-of-page link: the Kimi Coding Plan aff link (was the bare kimi.com domain)", () => { +test("kimi-web (Kimi Web) top-of-page link: points to www.kimi.ai", () => { const kimiWeb = providers.WEB_COOKIE_PROVIDERS["kimi-web"]; assert.ok(kimiWeb, "kimi-web must still exist in the web-cookie catalog"); assert.equal(kimiWeb.name, "Kimi Web", "display name is unchanged by the rename"); - assert.equal(kimiWeb.website, KIMI_CODING_AFF_URL); + assert.equal(kimiWeb.website, "https://www.kimi.ai"); }); test("kimi-coding-apikey (hidden, folds into kimi-coding card) also carries the aff link", () => { @@ -69,8 +68,6 @@ test("no visible Kimi provider website field still points at the unattributed pl test("runtime endpoints are untouched by the rename/aff-link changes (moonshot API base URL still api.moonshot.ai)", async () => { // Guard against the aff-link change ever leaking into a runtime executor // config — website is a UI navigation field only, never a fetch target. - const registry = await import( - "../../open-sse/config/providers/registry/moonshot/index.ts" - ); + const registry = await import("../../open-sse/config/providers/registry/moonshot/index.ts"); assert.equal(registry.moonshotProvider.baseUrl, "https://api.moonshot.ai/v1/chat/completions"); }); diff --git a/tests/unit/learned-reasoning-effort-caps.test.ts b/tests/unit/learned-reasoning-effort-caps.test.ts new file mode 100644 index 000000000000..5d69a342c38f --- /dev/null +++ b/tests/unit/learned-reasoning-effort-caps.test.ts @@ -0,0 +1,126 @@ +import { test, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { + REASONING_EFFORT_ORDER, + parseReasoningEffortEnum, + recordLearnedReasoningEffort, + getLearnedReasoningEffort, + __test_resetLearnedReasoningEffortCaps, +} from "../../open-sse/services/learnedReasoningEffortCaps.ts"; + +beforeEach(() => { + __test_resetLearnedReasoningEffortCaps(); +}); + +after(() => { + __test_resetLearnedReasoningEffortCaps(); +}); + +// ── REASONING_EFFORT_ORDER ────────────────────────────────────────────────── + +test("REASONING_EFFORT_ORDER is none < minimal < low < medium < high < xhigh < max", () => { + assert.deepEqual(REASONING_EFFORT_ORDER, [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + ]); +}); + +// ── parseReasoningEffortEnum ──────────────────────────────────────────────── + +test("parseReasoningEffortEnum extracts the real OVH 422 enum (backtick-quoted)", () => { + const err = + "Failed to deserialize the JSON body into the target type: reasoning_effort: " + + "unknown variant `xhigh`, expected one of `none`, `high`, `medium`, `low`, `minimal`"; + assert.deepEqual(parseReasoningEffortEnum(err), ["none", "high", "medium", "low", "minimal"]); +}); + +test("parseReasoningEffortEnum extracts a bare comma/and-joined enum with annotations", () => { + const err = + "Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low."; + assert.deepEqual(parseReasoningEffortEnum(err), ["xhigh", "medium", "low"]); +}); + +test("parseReasoningEffortEnum drops unrecognized tokens", () => { + const err = "expected one of `none`, `turbo`, `high`"; + assert.deepEqual(parseReasoningEffortEnum(err), ["none", "high"]); +}); + +test("parseReasoningEffortEnum returns null for unrelated error text", () => { + assert.equal(parseReasoningEffortEnum("connection refused"), null); + assert.equal(parseReasoningEffortEnum(""), null); + assert.equal(parseReasoningEffortEnum(null), null); + assert.equal(parseReasoningEffortEnum(undefined), null); +}); + +test("parseReasoningEffortEnum returns null when the list has no recognized token", () => { + assert.equal(parseReasoningEffortEnum("expected one of `foo`, `bar`"), null); +}); + +// ── recordLearnedReasoningEffort / getLearnedReasoningEffort ─────────────── + +test("records the highest recognized value from the accepted list", () => { + const learned = recordLearnedReasoningEffort("ovh", "qwen3-coder-30b-a3b-instruct", [ + "none", + "high", + "medium", + "low", + "minimal", + ]); + assert.equal(learned, "high"); + assert.equal(getLearnedReasoningEffort("ovh", "qwen3-coder-30b-a3b-instruct"), "high"); +}); + +test("returns null and stores nothing when acceptedValues has no recognized token", () => { + const learned = recordLearnedReasoningEffort("acme", "model-x", ["foo", "bar"]); + assert.equal(learned, null); + assert.equal(getLearnedReasoningEffort("acme", "model-x"), null); +}); + +test("monotonic decrease: a later, higher accepted-list never ratchets the cap back up", () => { + recordLearnedReasoningEffort("acme", "model-x", ["none", "low", "medium"]); + const learned = recordLearnedReasoningEffort("acme", "model-x", [ + "none", + "low", + "medium", + "high", + "xhigh", + ]); + assert.equal(learned, "medium"); + assert.equal(getLearnedReasoningEffort("acme", "model-x"), "medium"); +}); + +test("a later, lower accepted-list does ratchet the cap down", () => { + recordLearnedReasoningEffort("acme", "model-x", ["none", "low", "medium", "high"]); + const learned = recordLearnedReasoningEffort("acme", "model-x", ["none", "low"]); + assert.equal(learned, "low"); + assert.equal(getLearnedReasoningEffort("acme", "model-x"), "low"); +}); + +test("getLearnedReasoningEffort returns null for unknown provider+model", () => { + assert.equal(getLearnedReasoningEffort("acme", "unknown-model"), null); +}); + +test("getLearnedReasoningEffort is keyed case-insensitively on provider+model", () => { + recordLearnedReasoningEffort("OVH", "Qwen3-Coder-30B", ["none", "high"]); + assert.equal(getLearnedReasoningEffort("ovh", "qwen3-coder-30b"), "high"); + assert.equal(getLearnedReasoningEffort("OVH", "QWEN3-CODER-30B"), "high"); +}); + +test("different providers for the same model id have independent caps", () => { + recordLearnedReasoningEffort("ovh", "shared-model", ["none", "high"]); + assert.equal(getLearnedReasoningEffort("openrouter", "shared-model"), null); +}); + +test("handles empty/null provider or model gracefully", () => { + assert.equal(getLearnedReasoningEffort("", "m"), null); + assert.equal(getLearnedReasoningEffort("p", ""), null); + assert.equal(getLearnedReasoningEffort(null, "m"), null); + assert.equal(getLearnedReasoningEffort("p", null), null); + assert.equal(recordLearnedReasoningEffort("", "m", ["high"]), null); + assert.equal(recordLearnedReasoningEffort("p", "", ["high"]), null); +}); diff --git a/tests/unit/local-redis-status.test.ts b/tests/unit/local-redis-status.test.ts new file mode 100644 index 000000000000..956930664230 --- /dev/null +++ b/tests/unit/local-redis-status.test.ts @@ -0,0 +1,38 @@ +/** + * tests/unit/local-redis-status.test.ts + * + * Coverage for src/app/api/local/redis/status/route.ts: + * - The status endpoint must report OmniRoute as "connected" whenever the + * native REDIS_URL is reachable — not only when a Docker/Podman container + * is present. This is the production path used by this instance + * (redis on 127.0.0.1:6379, no container). + * + * Verified at the source-contract level (the route imports Next.js + the route + * guard, which is heavy to import in the native runner and would make the test + * environment-dependent on a live container runtime). + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const STATUS_SRC = path.resolve(__dirname, "../../src/app/api/local/redis/status/route.ts"); +const src = fs.readFileSync(STATUS_SRC, "utf8"); + +test("redis status: reports running via REDIS_URL reachability, not only Docker", () => { + assert.ok(src.includes("parseRedisUrl"), "status route must parse REDIS_URL"); + assert.ok( + src.includes("redisUrlReachable"), + "status route must probe REDIS_URL reachability" + ); + assert.ok( + src.includes("redisUrlConfigured"), + "status route must report whether REDIS_URL is configured" + ); + assert.ok( + src.includes("const running = container.running || redisUrlReachable;"), + "status route must treat a reachable native REDIS_URL as a connected state" + ); +}); diff --git a/tests/unit/local-rerank-logging.test.ts b/tests/unit/local-rerank-logging.test.ts new file mode 100644 index 000000000000..3c6ec6279458 --- /dev/null +++ b/tests/unit/local-rerank-logging.test.ts @@ -0,0 +1,215 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-rerank-test-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const { invalidateDbCache } = await import("../../src/lib/db/readCache.ts"); +const { createProviderNode, createProviderConnection } = + await import("../../src/lib/db/providers.ts"); +const { getCallLogs, getCallLogById, waitForCallLogSaves } = + await import("../../src/lib/usage/callLogs.ts"); +const { POST } = await import("../../src/app/api/v1/rerank/route.ts"); + +interface RerankSuccessResponse { + results: Array<{ index: number; relevance_score: number }>; +} + +interface CallLogRow { + id: string; + model: string; + provider: string; + status: number; + error?: string; + connectionId?: string; +} + +test.describe("Local rerank provider logging and fallback", () => { + const originalFetch = globalThis.fetch; + + test.after(() => { + globalThis.fetch = originalFetch; + core.resetDbInstance(); + try { + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + } catch { + // ignore + } + }); + + test("successfully logs local rerank calls and attaches metadata headers", async () => { + const now = new Date().toISOString(); + await createProviderNode({ + id: "vram", + name: "vram", + type: "openai", + prefix: "vram", + baseUrl: "http://127.0.0.1:8000/v1", + createdAt: now, + updatedAt: now, + }); + + await createProviderConnection({ + id: "conn-vram-1", + provider: "vram", + authType: "apikey", + name: "vram-local", + apiKey: "test-token", + createdAt: now, + updatedAt: now, + }); + + invalidateDbCache("nodes"); + invalidateDbCache("connections"); + + globalThis.fetch = async (url: string | URL | Request, init?: RequestInit) => { + assert.equal(String(url), "http://127.0.0.1:8000/v1/rerank"); + const parsedBody = JSON.parse(String(init?.body || "{}")); + assert.equal(parsedBody.model, "BAAI/bge-reranker-v2-m3"); + assert.equal(parsedBody.query, "test query"); + assert.deepEqual(parsedBody.documents, ["doc1", "doc2"]); + + return new Response( + JSON.stringify({ + results: [ + { index: 0, relevance_score: 0.95 }, + { index: 1, relevance_score: 0.2 }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + }; + + const req = new Request("http://localhost:20128/api/v1/rerank", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + model: "vram/BAAI/bge-reranker-v2-m3", + query: "test query", + documents: ["doc1", "doc2"], + }), + }); + + const res = await POST(req, {} as Record); + assert.equal(res.status, 200); + assert.equal(res.headers.get("x-omniroute-provider"), "vram"); + assert.equal(res.headers.get("x-omniroute-model"), "BAAI/bge-reranker-v2-m3"); + + const json = (await res.json()) as RerankSuccessResponse; + assert.equal(json.results.length, 2); + + await waitForCallLogSaves(15000); + + const logs = (await getCallLogs({ limit: 10 })) as unknown as CallLogRow[]; + const logEntry = logs.find((l) => l.model === "vram/BAAI/bge-reranker-v2-m3"); + assert.ok(logEntry, "Expected call log entry for local rerank"); + assert.equal(logEntry.provider, "vram"); + assert.equal(logEntry.status, 200); + + const detail = await getCallLogById(logEntry.id); + assert.deepEqual(detail?.requestBody, { + model: "vram/BAAI/bge-reranker-v2-m3", + query: "test query", + documents: ["doc1", "doc2"], + }); + assert.deepEqual(detail?.responseBody, { + results: [ + { index: 0, relevance_score: 0.95 }, + { index: 1, relevance_score: 0.2 }, + ], + }); + }); + + test("falls back from /v1/rerank to /rerank when local provider returns 404", async () => { + const now = new Date().toISOString(); + await createProviderNode({ + id: "infinity", + name: "infinity", + type: "openai", + prefix: "infinity", + baseUrl: "http://127.0.0.1:7997", + createdAt: now, + updatedAt: now, + }); + + await createProviderConnection({ + id: "conn-infinity-1", + provider: "infinity", + authType: "apikey", + name: "infinity-local", + apiKey: "test-token", + createdAt: now, + updatedAt: now, + }); + + invalidateDbCache("nodes"); + invalidateDbCache("connections"); + + const urlsAttempted: string[] = []; + globalThis.fetch = async (url: string | URL | Request) => { + urlsAttempted.push(String(url)); + if (String(url).endsWith("/v1/rerank")) { + return new Response("Not Found", { status: 404 }); + } + return new Response( + JSON.stringify({ + results: [{ index: 0, relevance_score: 0.99 }], + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + }; + + const req = new Request("http://localhost:20128/api/v1/rerank", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + model: "infinity/bge-reranker-large", + query: "search", + documents: ["doc1"], + }), + }); + + const res = await POST(req, {} as Record); + assert.equal(res.status, 200); + assert.deepEqual(urlsAttempted, [ + "http://127.0.0.1:7997/v1/rerank", + "http://127.0.0.1:7997/rerank", + ]); + }); + + test("records error call log when local provider returns 500", async () => { + globalThis.fetch = async () => { + return new Response(JSON.stringify({ detail: "Local backend failure" }), { + status: 500, + headers: { "Content-Type": "application/json" }, + }); + }; + + const req = new Request("http://localhost:20128/api/v1/rerank", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + model: "vram/BAAI/bge-reranker-v2-m3", + query: "test query", + documents: ["doc1"], + }), + }); + + const res = await POST(req, {} as Record); + assert.equal(res.status, 500); + + await waitForCallLogSaves(15000); + + const logs = (await getCallLogs({ limit: 10 })) as unknown as CallLogRow[]; + const logEntry = logs.find( + (l) => l.model === "vram/BAAI/bge-reranker-v2-m3" && l.status === 500 + ); + assert.ok(logEntry, "Expected 500 call log entry for local rerank failure"); + assert.equal(logEntry.provider, "vram"); + assert.equal(logEntry.error, "Local backend failure"); + }); +}); diff --git a/tests/unit/logfare-registry.test.ts b/tests/unit/logfare-registry.test.ts new file mode 100644 index 000000000000..d5439cd1b23e --- /dev/null +++ b/tests/unit/logfare-registry.test.ts @@ -0,0 +1,63 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { logfareProvider } from "../../open-sse/config/providers/registry/logfare/index.ts"; + +const { APIKEY_PROVIDERS } = await import( + "../../src/shared/constants/providers.ts" +); +const { REGISTRY: providerRegistry } = + await import("../../open-sse/config/providerRegistry.ts"); +const { NAMED_OPENAI_STYLE_PROVIDERS, isNamedOpenAIStyleProvider } = + await import( + "../../src/app/api/providers/[id]/models/discovery/providerSets.ts" + ); + +const SPEC = { + id: "logfare", + alias: "logfare", + name: "Logfare", + website: "https://logfare.ai", + chatUrl: "https://logfare.ai/v1/chat/completions", + modelsUrl: "https://logfare.ai/v1/models", +}; + +test("logfareProvider registry entry has correct configuration", () => { + assert.equal(logfareProvider.id, "logfare"); + assert.equal(logfareProvider.alias, "logfare"); + assert.equal(logfareProvider.format, "openai"); + assert.equal(logfareProvider.executor, "default"); + assert.equal(logfareProvider.baseUrl, SPEC.chatUrl); + assert.equal(logfareProvider.modelsUrl, SPEC.modelsUrl); + assert.equal(logfareProvider.authType, "apikey"); + assert.equal(logfareProvider.authHeader, "bearer"); + // Catalog is discovered live from /v1/models; no hardcoded seed. + assert.equal(logfareProvider.passthroughModels, true); + assert.equal(logfareProvider.models.length, 0); +}); + +test("APIKEY_PROVIDERS.logfare is registered with the canonical identity", () => { + const entry = APIKEY_PROVIDERS[SPEC.id]; + assert.ok(entry, `APIKEY_PROVIDERS.${SPEC.id} must be defined`); + assert.equal(entry.id, SPEC.id); + assert.equal(entry.alias, SPEC.alias); + assert.equal(entry.name, SPEC.name); + assert.equal(entry.website, SPEC.website); + assert.equal(entry.hasFree, true); + assert.equal(typeof entry.freeNote, "string"); + assert.equal(typeof entry.apiHint, "string"); + assert.match(entry.color, /^#[0-9A-Fa-f]{6}$/); +}); + +test("providerRegistry exposes the OpenAI-compatible chat completions URL", () => { + assert.equal(providerRegistry[SPEC.id].baseUrl, SPEC.chatUrl); + assert.equal(providerRegistry[SPEC.id].modelsUrl, SPEC.modelsUrl); +}); + +test("logfare is classified as a named OpenAI-style provider (live-fetch path)", () => { + assert.ok( + NAMED_OPENAI_STYLE_PROVIDERS.has(SPEC.id), + "logfare must be in NAMED_OPENAI_STYLE_PROVIDERS for live /v1/models fetch" + ); + assert.equal(isNamedOpenAIStyleProvider(SPEC.id), true); +}); diff --git a/tests/unit/memory-system-first-6135.test.ts b/tests/unit/memory-system-first-6135.test.ts index 7104007ae641..339f83792cf0 100644 --- a/tests/unit/memory-system-first-6135.test.ts +++ b/tests/unit/memory-system-first-6135.test.ts @@ -49,6 +49,11 @@ describe("injectMemory system-must-be-first (#6135)", () => { it("flags xiaomi-mimo (and alias mimo) as system-must-be-first", () => { assert.equal(systemMessageMustBeFirst("xiaomi-mimo"), true); assert.equal(systemMessageMustBeFirst("mimo"), true); + // tokenrouter: confirmed live 2026-08-22 — mid-array system message + // (e.g. the purifyHistory compression notice) -> HTTP 400 + // "System message must be at the beginning". + assert.equal(systemMessageMustBeFirst("tokenrouter"), true); + assert.equal(systemMessageMustBeFirst("TokenRouter"), true); // case-insensitive // default: unlisted providers keep current (non-first-constrained) behavior assert.equal(systemMessageMustBeFirst("anthropic"), false); assert.equal(systemMessageMustBeFirst(null), false); diff --git a/tests/unit/oauth-device-flow-11164.test.ts b/tests/unit/oauth-device-flow-11164.test.ts new file mode 100644 index 000000000000..48cdb23b6406 --- /dev/null +++ b/tests/unit/oauth-device-flow-11164.test.ts @@ -0,0 +1,38 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; + +test("device code normalization handles camelCase, snake_case, and authUrl without returning undefined", () => { + const cases = [ + { + input: { userCode: "ABCD-1234", verificationUri: "https://auth.example.com" }, + expectedCode: "ABCD-1234", + expectedUri: "https://auth.example.com", + }, + { + input: { user_code: "EFGH-5678", verification_uri: "https://auth.example.com/device" }, + expectedCode: "EFGH-5678", + expectedUri: "https://auth.example.com/device", + }, + { + input: { authUrl: "https://studio.example.com/auth" }, + expectedCode: "", + expectedUri: "https://studio.example.com/auth", + }, + ]; + + for (const c of cases) { + const userCode = c.input.userCode ?? c.input.user_code ?? ""; + const verificationUri = + c.input.verificationUriComplete ?? + c.input.verification_uri_complete ?? + c.input.verificationUri ?? + c.input.verification_uri ?? + c.input.authUrl ?? + c.input.url ?? + ""; + + assert.equal(userCode, c.expectedCode); + assert.equal(verificationUri, c.expectedUri); + assert.notEqual(verificationUri, "undefined"); + } +}); diff --git a/tests/unit/ollama-404-model-lockout-11071.test.ts b/tests/unit/ollama-404-model-lockout-11071.test.ts new file mode 100644 index 000000000000..f58c29c07b97 --- /dev/null +++ b/tests/unit/ollama-404-model-lockout-11071.test.ts @@ -0,0 +1,66 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-ollama-404-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const auth = await import("../../src/sse/services/auth.ts"); +const { hasPerModelQuota, isModelLocked } = await import("../../open-sse/services/accountFallback.ts"); + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("hasPerModelQuota returns true for ollama-local and ollama providers", () => { + assert.equal(hasPerModelQuota("ollama-local"), true); + assert.equal(hasPerModelQuota("ollama"), true); +}); + +test("markAccountUnavailable locks only the missing model on a 404 from ollama-local", async () => { + await resetStorage(); + + const connection = await providersDb.createProviderConnection({ + provider: "ollama-local", + authType: "none", + baseUrl: "http://127.0.0.1:11434/v1", + isActive: true, + }); + + const result = await auth.markAccountUnavailable( + connection.id, + 404, + "model 'model-b' not found", + "ollama-local", + "model-b" + ); + + assert.equal(result.shouldFallback, true); + + // The missing model must be locked + assert.equal(isModelLocked("ollama-local", connection.id, "model-b"), true); + + // The connection in DB must remain active / not marked unavailable for sibling models + const connInDb = await providersDb.getProviderConnectionById(connection.id); + assert.notEqual(connInDb?.testStatus, "unavailable", "connection should not be marked unavailable connection-wide on a 404 model-not-found error"); + + // getProviderCredentials must still serve sibling models + const selectedForSibling = await auth.getProviderCredentials( + "ollama-local", + null, + null, + "model-a" + ); + assert.ok(selectedForSibling && !("allExpired" in selectedForSibling), "sibling model-a must still be selected on the same connection"); +}); diff --git a/tests/unit/opencode-empty-rejection-rotation.test.ts b/tests/unit/opencode-empty-rejection-rotation.test.ts new file mode 100644 index 000000000000..784a4b83a747 --- /dev/null +++ b/tests/unit/opencode-empty-rejection-rotation.test.ts @@ -0,0 +1,416 @@ +import { describe, it, beforeEach, afterEach, before, after } from "node:test"; +import assert from "node:assert"; +import net from "node:net"; +import { OpencodeExecutor } from "../../open-sse/executors/opencode.ts"; +import type { ExecutorLog, ProviderCredentials } from "../../open-sse/executors/base.ts"; +import { resolveProxyForRequest } from "../../open-sse/utils/proxyFetch.ts"; +import { + isEmptyUpstreamRejection, + extractChatcmplId, +} from "../../open-sse/executors/accountRotation.ts"; + +/** + * Empty-upstream-rejection rotation (#design opencode-empty-rejection-rotation). + * + * An upstream 400 whose body carries no usable completion (the observed malformed + * envelope: `choices[0].message` with no error field, no real content, + * `finish_reason: null`) must be rotated/retried instead of propagated as a fatal + * success — that was killing subagent sessions. These tests pin the wiring: + * + * 1. A 400 empty rejection rotates to the next account (and its proxy). + * 2. The retry budget is bounded: +1 attempt for a single account, exactly N + * for an N-account all-empty run (propagate the last 400, never loop forever). + * 3. A 400 carrying a real error field (or non-empty content) still propagates + * immediately — no cooldown, no success, no rotation. + * 4. The 200/success path is never cloned or read (anti-bufferisation). + * + * The dispatch layer is mocked by stubbing globalThis.fetch (exactly what the + * #4954 proxy integration test does). Three throwaway TCP listeners stand in for + * the per-account proxies so runWithProxyContext's reachability probe passes. + */ + +const log: ExecutorLog = { debug() {}, info() {}, warn() {}, error() {} }; + +const ACCOUNT_A = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; +const ACCOUNT_B = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; +const ACCOUNT_C = "cccccccccccccccccccccccccccccccc"; + +const EMPTY_BODY = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; +const ERROR_BODY = JSON.stringify({ + error: { message: "bad request", type: "invalid_request_error" }, +}); + +let serverA: net.Server; +let serverB: net.Server; +let serverC: net.Server; +let portA = 0; +let portB = 0; +let portC = 0; + +function listen(server: net.Server): Promise { + return new Promise((resolve) => { + server.listen(0, "127.0.0.1", () => { + resolve((server.address() as net.AddressInfo).port); + }); + }); +} + +before(async () => { + serverA = net.createServer((s) => s.destroy()); + serverB = net.createServer((s) => s.destroy()); + serverC = net.createServer((s) => s.destroy()); + portA = await listen(serverA); + portB = await listen(serverB); + portC = await listen(serverC); +}); + +after(() => { + serverA?.close(); + serverB?.close(); + serverC?.close(); +}); + +function portFor(fp: string): number { + if (fp === ACCOUNT_A) return portA; + if (fp === ACCOUNT_B) return portB; + return portC; +} + +/** `fingerprints` accounts; `proxied` is the subset that get a dedicated proxy + * (defaults to all). A proxy-less account shares the default egress. */ +function credentialsFor( + fingerprints: string[], + proxied: string[] = [...fingerprints] +): ProviderCredentials { + return { + apiKey: null, + accessToken: null, + connectionId: "noauth", + providerSpecificData: { + fingerprints, + ...(proxied.length > 0 && { + accountProxies: proxied.map((fp) => ({ + fingerprint: fp, + proxy: { type: "http", host: "127.0.0.1", port: portFor(fp) }, + })), + }), + }, + }; +} + +/** A Response subclass that counts clone() so we can assert the executor never + * buffers a 200/streaming response. Note: `clone()` returns a plain Response, so + * only `clone()` is reliably counted (a read on the clone hits the native + * method, not this override) — counting clones is the meaningful invariant. */ +class SpyResponse extends Response { + static clones = 0; + clone(): Response { + SpyResponse.clones++; + return super.clone(); + } +} + +interface PlanStep { + status: number; + body?: string; + throw?: Error; +} + +describe("OpencodeExecutor empty-rejection rotation", () => { + let originalFetch: typeof globalThis.fetch; + let observed: Array<{ source: string; host: string | null; port: string | null }>; + const GUARD_FLAG = "NETWORK_ROTATION_SHARED_EGRESS_GUARD"; + let savedGuardFlag: string | undefined; + + beforeEach(() => { + originalFetch = globalThis.fetch; + observed = []; + SpyResponse.clones = 0; + savedGuardFlag = process.env[GUARD_FLAG]; + delete process.env[GUARD_FLAG]; + }); + + afterEach(() => { + globalThis.fetch = originalFetch; + if (savedGuardFlag === undefined) delete process.env[GUARD_FLAG]; + else process.env[GUARD_FLAG] = savedGuardFlag; + }); + + function installFetch(plan: PlanStep[]) { + let call = 0; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = + typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + const resolved = resolveProxyForRequest(url); + observed.push({ + source: resolved.source, + host: resolved.proxyUrl ? new URL(resolved.proxyUrl).hostname : null, + port: resolved.proxyUrl ? new URL(resolved.proxyUrl).port : null, + }); + const step = plan[Math.min(call, plan.length - 1)]; + call++; + if (step.throw) throw step.throw; + return new SpyResponse(step.body ?? JSON.stringify({ ok: step.status === 200 }), { + status: step.status, + headers: { "Content-Type": "application/json" }, + }); + }) as typeof globalThis.fetch; + } + + /** + * Launches the executor. Asserts the predicate itself behaves (regression guard + * for the design's signature — the wiring tests below depend on it). + */ + it("predicate matches the observed envelope and rejects real errors", () => { + assert.strictEqual(isEmptyUpstreamRejection(400, EMPTY_BODY), true); + assert.strictEqual(isEmptyUpstreamRejection(200, EMPTY_BODY), false); + assert.strictEqual(isEmptyUpstreamRejection(400, ERROR_BODY), false); + assert.strictEqual(extractChatcmplId(EMPTY_BODY), "chatcmpl_44fn2g6e7kk"); + }); + + it("rotates to the next account on an empty 400 rejection (loop)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 200, + "must rotate past the empty 400" + ); + assert.ok(observed.length >= 2, "should have dispatched on a second account"); + assert.ok( + observed.some((o) => o.port === String(portA)), + "first attempt on account A" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "rotated attempt on account B" + ); + }); + + it("caps an all-empty N-account run at N attempts and propagates the last 400 intact", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + { status: 200 }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 400, + "must propagate the last empty 400" + ); + assert.strictEqual(observed.length, 3, "must NOT exceed N attempts (no infinite loop)"); + assert.ok(SpyResponse.clones >= 1, "the empty 400 path must read the body to classify it"); + const propagated = await (result as { response: Response }).response.clone().text(); + assert.strictEqual(propagated, EMPTY_BODY, "propagated 400 body must stay intact"); + for (const p of observed) { + assert.strictEqual(p.source, "context", "every dispatch must egress through a proxy context"); + } + }); + + it("retries the same proxied account once when it is the only account", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A]), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + assert.strictEqual(observed.length, 2, "exactly one bounded retry on the sole account"); + assert.ok( + observed.every((o) => o.port === String(portA)), + "both attempts egress through the single account's proxy" + ); + }); + + it("coexists with 429 rotation and 200 success in the same request", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 429 }, { status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 200, + "final response should succeed" + ); + assert.strictEqual(observed.length, 3, "429 + empty-400 + success across three accounts"); + assert.ok( + observed.some((o) => o.port === String(portA)), + "account A (429)" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "account B (empty 400)" + ); + assert.ok( + observed.some((o) => o.port === String(portC)), + "account C (200)" + ); + }); + + it("propagates a 400 carrying an error field immediately (no rotation)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: ERROR_BODY }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 400, + "real error 400 must propagate" + ); + assert.strictEqual(observed.length, 1, "must NOT rotate on a genuine error 400"); + }); + + it("never clones or reads the body of a 200 via the loop", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 200 }, { status: 200 }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 200); + assert.strictEqual(SpyResponse.clones, 0, "loop 200 must never be cloned"); + }); + + it("retries once via the fast path when a direct account answers an empty 400", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + assert.strictEqual(observed.length, 2, "fast path must retry the direct account exactly once"); + }); + + it("propagates the second 400 intact when the fast path retries and empty-rejects again", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + const propagated = await (result as { response: Response }).response.clone().text(); + assert.strictEqual(propagated, EMPTY_BODY, "second rejection propagates with intact body"); + assert.strictEqual(observed.length, 2, "exactly one retry, no loop"); + }); + + it("never clones or reads the body of a 200 via the fast path", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 200); + assert.strictEqual(SpyResponse.clones, 0, "fast path 200 must never be cloned"); + }); + + it("rotates to a proxied account after a proxy-less account empty-rejects (shared-egress guard on by default)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + // A proxy-less, B proxied: B must still be tried and succeed. + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B], [ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: { status: number } }).response.status, + 200, + "the proxied account (B) must still be tried and must succeed" + ); + assert.strictEqual(observed.length, 2, "exactly one empty rejection (A) then one success (B)"); + assert.ok( + observed.some((o) => o.source === "direct"), + "first dispatch on the proxy-less account" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "rotated dispatch on the proxied account" + ); + }); +}); diff --git a/tests/unit/opencode-v2-config-11070.test.ts b/tests/unit/opencode-v2-config-11070.test.ts new file mode 100644 index 000000000000..71bd8a953781 --- /dev/null +++ b/tests/unit/opencode-v2-config-11070.test.ts @@ -0,0 +1,40 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const opencodeConfig = await import("../../src/shared/services/opencodeConfig.ts"); + +test("buildOpenCodeConfigDocument includes both V1 (provider) and V2 (providers) definitions", () => { + const doc = opencodeConfig.buildOpenCodeConfigDocument({ + baseUrl: "http://localhost:20128/v1", + apiKey: "{env:OMNIROUTE_API_KEY}", + models: ["auto/best-coding"], + }); + + assert.ok(doc.provider?.omniroute, "V1 provider.omniroute must be present"); + assert.equal(doc.provider.omniroute.npm, "@ai-sdk/openai-compatible"); + assert.equal(doc.provider.omniroute.options.baseURL, "http://localhost:20128/v1"); + + assert.ok(doc.providers?.omniroute, "V2 providers.omniroute must be present"); + assert.equal(doc.providers.omniroute.package, "@opencode-ai/ai/providers/openai-compatible"); + assert.equal(doc.providers.omniroute.settings.baseURL, "http://localhost:20128/v1"); + assert.equal(doc.providers.omniroute.settings.apiKey, "{env:OMNIROUTE_API_KEY}"); + assert.ok(doc.providers.omniroute.models["auto/best-coding"].limit, "V2 model limit must be present"); +}); + +test("mergeOpenCodeConfig preserves existing properties and updates both provider and providers", () => { + const existing = { + $schema: "https://opencode.ai/config.json", + customField: "keep-me", + }; + + const merged = opencodeConfig.mergeOpenCodeConfig(existing, { + baseUrl: "http://localhost:20128/v1", + apiKey: "sk_test_key", + models: ["auto/best-coding"], + }); + + assert.equal(merged.customField, "keep-me"); + assert.ok(merged.provider?.omniroute); + assert.ok(merged.providers?.omniroute); + assert.equal(merged.providers.omniroute.settings.apiKey, "sk_test_key"); +}); diff --git a/tests/unit/opencode-zen-go-shared-models.test.ts b/tests/unit/opencode-zen-go-shared-models.test.ts index 8c3079ad1556..c1fb49e97d5f 100644 --- a/tests/unit/opencode-zen-go-shared-models.test.ts +++ b/tests/unit/opencode-zen-go-shared-models.test.ts @@ -24,3 +24,16 @@ test("every OPENCODE_ZEN_GO_SHARED_MODELS entry is present, unmodified, exactly test("OPENCODE_ZEN_GO_SHARED_MODELS is frozen (no accidental cross-registry mutation)", () => { assert.ok(Object.isFrozen(OPENCODE_ZEN_GO_SHARED_MODELS)); }); + +test("referenced non-shared model ids remain present", () => { + const goIds = new Set(opencode_goProvider.models.map((m) => m.id)); + const zenIds = new Set(opencode_zenProvider.models.map((m) => m.id)); + for (const id of ["minimax-m3", "glm-5.1"]) { + assert.ok(goIds.has(id) || zenIds.has(id), `expected ${id} in go or zen`); + } +}); + +test("models[0] is the intended dashboard default", () => { + assert.equal(opencode_goProvider.models[0].id, "glm-5.2"); + assert.equal(opencode_zenProvider.models[0].id, "big-pickle"); +}); diff --git a/tests/unit/perplexity-discovery-filter.test.ts b/tests/unit/perplexity-discovery-filter.test.ts new file mode 100644 index 000000000000..853364674651 --- /dev/null +++ b/tests/unit/perplexity-discovery-filter.test.ts @@ -0,0 +1,63 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { PROVIDER_MODELS_CONFIG } from "../../src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts"; + +// Regression guard for #11060 — Perplexity's /v1/models endpoint lists the +// Agent API catalog (vendor-prefixed ids like "anthropic/claude-fable-5"), but +// chat requests always go to the classic /chat/completions endpoint, which only +// accepts the Sonar family. Without a PROVIDER_MODELS_CONFIG entry, generic +// model import pulled those agent-style ids into the connection's chat model +// list and every routed request failed with 400 "Invalid model". The discovery +// entry must exist and its parseResponse must keep only Sonar-family ids. + +test("perplexity has a discovery entry in PROVIDER_MODELS_CONFIG", () => { + const cfg = PROVIDER_MODELS_CONFIG.perplexity; + assert.ok(cfg, "expected a perplexity entry in PROVIDER_MODELS_CONFIG"); + assert.equal(cfg.method, "GET"); + assert.equal(cfg.url, "https://api.perplexity.ai/v1/models"); + assert.equal(typeof cfg.parseResponse, "function"); +}); + +test("perplexity parseResponse keeps only the Sonar family (#11060)", () => { + const cfg = PROVIDER_MODELS_CONFIG.perplexity; + const models = cfg.parseResponse({ + object: "list", + data: [ + { id: "anthropic/claude-fable-5", object: "model", owned_by: "anthropic" }, + { id: "sonar-pro", object: "model", owned_by: "perplexity" }, + { id: "sonar", object: "model", owned_by: "perplexity" }, + ], + }) as Array<{ id: string }>; + + assert.deepEqual( + models.map((model) => model.id), + ["sonar-pro", "sonar"] + ); +}); + +test("perplexity parseResponse keeps every Sonar variant and drops non-Sonar ids", () => { + const cfg = PROVIDER_MODELS_CONFIG.perplexity; + const models = cfg.parseResponse({ + data: [ + { id: "sonar-deep-research" }, + { id: "sonar-reasoning-pro" }, + { id: "sonar-pro" }, + { id: "sonar" }, + { id: "openai/gpt-5" }, + { id: "sonarish" }, + ], + }) as Array<{ id: string }>; + + assert.deepEqual( + models.map((model) => model.id), + ["sonar-deep-research", "sonar-reasoning-pro", "sonar-pro", "sonar"] + ); +}); + +test("perplexity parseResponse tolerates empty and malformed payloads", () => { + const cfg = PROVIDER_MODELS_CONFIG.perplexity; + assert.deepEqual(cfg.parseResponse({ data: [] }), []); + assert.deepEqual(cfg.parseResponse(undefined), []); + assert.deepEqual(cfg.parseResponse({}), []); +}); diff --git a/tests/unit/pollinations-api-key-required-11096.test.ts b/tests/unit/pollinations-api-key-required-11096.test.ts new file mode 100644 index 000000000000..8b69fb833121 --- /dev/null +++ b/tests/unit/pollinations-api-key-required-11096.test.ts @@ -0,0 +1,11 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { providerAllowsOptionalApiKey } from "../../src/shared/constants/providers.js"; + +test("pollinations provider requires an API key and does not allow optional API key", () => { + assert.equal( + providerAllowsOptionalApiKey("pollinations"), + false, + "pollinations must require an API key because anonymous completions are no longer supported" + ); +}); diff --git a/tests/unit/private-host-ip-parity-11122.test.ts b/tests/unit/private-host-ip-parity-11122.test.ts new file mode 100644 index 000000000000..3fb4f1049012 --- /dev/null +++ b/tests/unit/private-host-ip-parity-11122.test.ts @@ -0,0 +1,135 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { isIP } from "node:net"; +import { fileURLToPath } from "node:url"; + +import { build } from "esbuild"; + +import { ipVersion, isPrivateHost } from "../../src/shared/network/privateHost.ts"; + +// #11122 — that PR pointed `isLocalProvider()` at `isPrivateHost`, imported from +// `outboundUrlGuard.ts` (which imports `node:net`). `open-sse/config/providerRegistry.ts` is in +// the `ProviderDetailPageClient.tsx` graph, so the browser bundle broke and +// media-page-client-browser-bundle.test.ts went red on release/v3.8.50. `isPrivateHost` moved +// here to fix it; two things must hold for that move to be safe: +// 1. `ipVersion` agrees with `node:net#isIP` on every input — a NARROWER match would classify +// a private address as public and open the egress the guard exists to close. +// 2. The module stays bundleable for the browser (no `node:*`, no `@/` alias). + +const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)); + +const LITERALS = [ + // IPv4 — valid + "0.0.0.0", + "127.0.0.1", + "10.0.0.1", + "100.64.0.1", + "169.254.169.254", + "172.16.0.1", + "172.31.255.254", + "192.168.1.50", + "8.8.8.8", + "255.255.255.255", + // IPv4 — invalid spellings node rejects + "010.1.1.1", + "1.2.3.4.5", + "1.2.3", + "256.1.1.1", + "1.2.3.-1", + "1.2.3.4 ", + " 1.2.3.4", + "1.2.3.04", + // IPv6 — valid + "::", + "::1", + "fd00::1", + "fe80::1", + "fc00::abcd", + "2001:db8::1", + "2001:0db8:0000:0000:0000:0000:0000:0001", + "::ffff:192.168.1.1", + "::ffff:a9fe:a9fe", + "64:ff9b::8.8.8.8", + "fe80::1%eth0", + "fe80::1%25", + // IPv6 — invalid + ":::", + "2001:db8::1::2", + "fe80::1%", + "gggg::1", + "2001:db8:::1", + // not IP literals at all + "", + "localhost", + "studio.local", + "api.openai.com", + "0x7f.1", + "2130706433", + "..", + "999", +]; + +test("ipVersion matches node:net#isIP across IP literals and near-misses", () => { + for (const host of LITERALS) { + assert.equal( + ipVersion(host), + isIP(host), + `ipVersion disagreed with isIP for ${JSON.stringify(host)}` + ); + } +}); + +test("ipVersion matches node:net#isIP across generated IPv4 permutations", () => { + const segments = ["0", "00", "01", "9", "10", "099", "127", "192", "255", "256", "300", ""]; + for (const a of segments) { + for (const b of segments) { + const host = `${a}.${b}.${a}.${b}`; + assert.equal(ipVersion(host), isIP(host), `ipVersion disagreed with isIP for ${host}`); + } + } +}); + +test("ipVersion matches node:net#isIP across generated IPv6 permutations", () => { + const groups = ["", "0", "1", "abcd", "ffff", "fffff", "xyz"]; + for (const g of groups) { + for (const host of [`${g}::1`, `::${g}`, `${g}:${g}::${g}`, `2001:db8::${g}`, `[${g}::1]`]) { + assert.equal(ipVersion(host), isIP(host), `ipVersion disagreed with isIP for ${host}`); + } + } +}); + +test("an over-long input is rejected rather than fed to the alternation", () => { + // The length guard is the ReDoS bound (AGENTS.md → "Regex Security"). Node agrees: no legal + // literal is this long, so the fast path costs no accuracy. + const long = `${"f".repeat(200)}::1`; + assert.equal(ipVersion(long), 0); + assert.equal(isIP(long), 0); +}); + +test("isPrivateHost keeps its verdicts after the move", () => { + for (const host of ["", "localhost", "127.0.0.1", "::1", "[::1]", "10.1.2.3", "192.168.0.15"]) { + assert.equal(isPrivateHost(host), true, `expected private: ${JSON.stringify(host)}`); + } + for (const host of ["api.openai.com", "8.8.8.8", "172.32.0.1", "2001:db8::1"]) { + assert.equal(isPrivateHost(host), false, `expected public: ${host}`); + } +}); + +test("privateHost stays browser-bundle safe", async () => { + // The direct guard for the regression: providerRegistry -> privateHost is in the + // ProviderDetailPageClient graph, so a `node:*` import here breaks the dashboard build. + await assert.doesNotReject( + build({ + absWorkingDir: REPO_ROOT, + entryPoints: [ + fileURLToPath(new URL("../../src/shared/network/privateHost.ts", import.meta.url)), + ], + bundle: true, + format: "esm", + logLevel: "silent", + platform: "browser", + tsconfig: "tsconfig.json", + write: false, + }) + ); +}); diff --git a/tests/unit/provider-connections-quota-threshold.test.ts b/tests/unit/provider-connections-quota-threshold.test.ts index 5caaeea36993..e7812557e994 100644 --- a/tests/unit/provider-connections-quota-threshold.test.ts +++ b/tests/unit/provider-connections-quota-threshold.test.ts @@ -106,18 +106,36 @@ test("updateProviderConnection with explicit null clears the column entirely", a assert.ok(reread.quotaWindowThresholds === null || reread.quotaWindowThresholds === undefined); }); -test("DB serializer drops out-of-range values silently", async () => { - // The DB module sanitizes the map on the way in; values outside 0-100 or - // non-integers are pruned. This is a defense in depth — the Zod schema - // already rejects them at the API boundary, but the DB shouldn't trust. - const created = await providersDb.createProviderConnection({ - provider: "codex", - authType: "apikey", - name: "Codex Sanitize", - apiKey: "sk-san", - quotaWindowThresholds: { window5h: 95, bogus: 999, fractional: 1.5 }, - }); - assert.deepEqual(created.quotaWindowThresholds, { window5h: 95 }); +test("DB serializer refuses out-of-range / invalid values instead of dropping silently", async () => { + // the DB module must refuse the write (throw) rather than + // silently prune invalid keys/values on the way in, so operator intent is + // never lost without an error. The Zod schema already rejects at the API + // boundary; this is defense in depth for direct DB writers (seed/scripts). + await assert.rejects( + () => + providersDb.createProviderConnection({ + provider: "codex", + authType: "apikey", + name: "Codex Sanitize", + apiKey: "sk-san", + quotaWindowThresholds: { window5h: 95, bogus: 999, fractional: 1.5 }, + }), + /rejected keys/ + ); +}); + +test("DB serializer refuses unknown rate-limit override keys instead of dropping silently", async () => { + await assert.rejects( + () => + providersDb.createProviderConnection({ + provider: "codex", + authType: "apikey", + name: "Codex Sanitize RLO", + apiKey: "sk-san-2", + rateLimitOverrides: { rpm: 10, bogus: 999, tpm: -1 }, + }), + /rejected keys/ + ); }); test("updateProviderConnectionSchema accepts a valid window map", () => { diff --git a/tests/unit/provider-error-rules-operator.test.ts b/tests/unit/provider-error-rules-operator.test.ts new file mode 100644 index 000000000000..6f8c9e52ef81 --- /dev/null +++ b/tests/unit/provider-error-rules-operator.test.ts @@ -0,0 +1,135 @@ +import { describe, it, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { + getProviderErrorRuleMatch, + setOperatorProviderErrorRules, + resolveRuleMatchBody, + honorsRuleLockScope, + type OperatorProviderErrorRule, +} from "../../open-sse/config/providerErrorRules.ts"; + +describe("operator error rules", () => { + beforeEach(() => { + // Isolate each test from the settings-backed cache. + setOperatorProviderErrorRules(undefined); + }); + + it("operator rule overrides the catalog registry for a provider", () => { + const op: Record = { + nvidia: [{ status: 404, match: "Not found for account", scope: "model", cooldownMs: 1000 }], + }; + const m = getProviderErrorRuleMatch("nvidia", 404, null, "Not found for account id 123", op); + assert.ok(m, "operator rule should match"); + assert.equal(m.scope, "model"); + assert.equal(m.cooldownMs, 1000); + }); + + it("operator rule wins even when a catalog rule would also match", () => { + const op: Record = { + openrouter: [{ status: 402, match: "credits exhausted", scope: "model" }], + }; + const m = getProviderErrorRuleMatch("openrouter", 402, null, "credits exhausted on key", op); + assert.ok(m); + // Catalog rule for openrouter/402 uses scope "connection"; the operator + // override must take precedence. + assert.equal(m.scope, "model"); + }); + + it("operator can reclassify a 401 before the global permanent rule", () => { + const op: Record = { + acme: [ + { status: 401, match: "transient quota", scope: "connection", reason: "quota_exhausted" }, + ], + }; + const m = getProviderErrorRuleMatch("acme", 401, null, "transient quota — retry shortly", op); + assert.ok(m); + assert.equal(m.scope, "connection"); + assert.equal(m.reason, "quota_exhausted"); + }); + + it("unknown provider with no operator rule returns null (no throw)", () => { + const m = getProviderErrorRuleMatch("unknown-provider", 402, null, "anything"); + assert.equal(m, null); + }); + + it("substring match is case-insensitive", () => { + const op: Record = { + nvidia: [{ status: 404, match: "NOT FOUND", scope: "model" }], + }; + const m = getProviderErrorRuleMatch("nvidia", 404, null, "Body says Not Found Here", op); + assert.ok(m); + assert.equal(m.scope, "model"); + }); + + it("status must match before the substring is considered", () => { + const op: Record = { + nvidia: [{ status: 404, match: "not found", scope: "model" }], + }; + // 500 with the same body text must NOT match a 404 rule. + const m = getProviderErrorRuleMatch("nvidia", 500, null, "not found for account", op); + assert.equal(m, null); + }); + + it("without an operator override the catalog registry is intact", () => { + const m = getProviderErrorRuleMatch("openrouter", 402, null, "credits exhausted on key"); + assert.ok(m); + assert.equal(m.scope, "connection"); + assert.equal(m.cooldownMs, 2 * 60 * 1000); + }); + + it("reads the settings-backed cache via setOperatorProviderErrorRules", () => { + setOperatorProviderErrorRules({ + nvidia: [{ status: 404, match: "Not found", scope: "model" }], + }); + const m = getProviderErrorRuleMatch("nvidia", 404, null, "Not found for account"); + assert.ok(m); + assert.equal(m.scope, "model"); + // Provider key lookup is case-insensitive. + const m2 = getProviderErrorRuleMatch("NVIDIA", 404, null, "Not found here"); + assert.ok(m2); + assert.equal(m2.scope, "model"); + }); + + // Regression coverage for #11104's original gap: an operator rule for any + // provider outside the built-in FULL_TEXT_RULE_PROVIDERS/ + // HONORS_RULE_LOCK_SCOPE_PROVIDERS allowlists was silently text-blind (only + // {code,type} reached the matcher) and had its declared scope dropped by the + // persistence layer. Declaring an operator rule for a provider must be + // sufficient by itself — no separate allowlist entry required. + describe("operator rule bypasses the built-in allowlists", () => { + it("resolveRuleMatchBody hands the full error text once an operator rule exists for the provider", () => { + setOperatorProviderErrorRules({ + acme: [{ status: 404, match: "model withdrawn", scope: "model" }], + }); + const body = resolveRuleMatchBody("acme", { code: "not_found" }, "Model withdrawn upstream"); + assert.equal(body, "Model withdrawn upstream"); + }); + + it("resolveRuleMatchBody keeps returning the structured error for a provider with no operator rule", () => { + const body = resolveRuleMatchBody("acme", { code: "not_found" }, "Model withdrawn upstream"); + assert.deepEqual(body, { code: "not_found" }); + }); + + it("honorsRuleLockScope is true once an operator rule exists for the provider", () => { + assert.equal(honorsRuleLockScope("acme"), false); + setOperatorProviderErrorRules({ + acme: [{ status: 404, match: "model withdrawn", scope: "model" }], + }); + assert.equal(honorsRuleLockScope("acme"), true); + }); + + it("an operator rule for a non-allowlisted provider matches on raw body text end to end", () => { + setOperatorProviderErrorRules({ + acme: [{ status: 404, match: "model withdrawn", scope: "model" }], + }); + const body = resolveRuleMatchBody( + "acme", + { code: "not_found" }, + "Error: model withdrawn upstream" + ); + const m = getProviderErrorRuleMatch("acme", 404, null, body); + assert.ok(m, "operator rule should match once resolveRuleMatchBody hands it the raw text"); + assert.equal(m.scope, "model"); + }); + }); +}); diff --git a/tests/unit/provider-route-schemas.test.ts b/tests/unit/provider-route-schemas.test.ts index ca61c4f28d0c..30b5d4bd8c26 100644 --- a/tests/unit/provider-route-schemas.test.ts +++ b/tests/unit/provider-route-schemas.test.ts @@ -5,17 +5,19 @@ const { createProviderSchema, providersBatchTestSchema } = await import("../../src/shared/validation/schemas.ts"); const { providerAllowsOptionalApiKey } = await import("../../src/shared/constants/providers.ts"); -test("Pollinations is treated as a keyless-capable provider", () => { - assert.equal(providerAllowsOptionalApiKey("pollinations"), true); +// #11117: Pollinations no longer serves anonymous requests (401 without a key), +// so it left EXPLICIT_OPTIONAL_APIKEY_PROVIDER_IDS — key is now required. +test("Pollinations requires an API key", () => { + assert.equal(providerAllowsOptionalApiKey("pollinations"), false); }); -test("createProviderSchema allows Pollinations without apiKey", () => { +test("createProviderSchema rejects Pollinations without apiKey", () => { const result = createProviderSchema.safeParse({ provider: "pollinations", name: "Pollinations", }); - assert.equal(result.success, true); + assert.equal(result.success, false); }); test("providersBatchTestSchema accepts cloud-agent batch mode", () => { diff --git a/tests/unit/providers-constants-split.test.ts b/tests/unit/providers-constants-split.test.ts index 3eabac24bb96..e22b3fd4e835 100644 --- a/tests/unit/providers-constants-split.test.ts +++ b/tests/unit/providers-constants-split.test.ts @@ -25,7 +25,7 @@ // #10729) brings it to 229; Token Kiosk (gateways, #10722) — merged in the same // merge-train batch — independently bumped the gateways family too, landing at 231; Freebuff // (gateways, #10531) brings it to 232. #8864 moves uncloseai (gateways family) into -// NOAUTH_PROVIDERS, dropping the APIKEY_PROVIDERS count to 231. +// NOAUTH_PROVIDERS, dropping the APIKEY_PROVIDERS count to 231. Logfare (gateways, #10987) brings it back to 232. import { test } from "node:test"; import assert from "node:assert/strict"; @@ -54,12 +54,12 @@ test("barrel still exports every catalog + key helpers", () => { } }); -test("APIKEY_PROVIDERS merges the 6 family files into 231 entries (no loss / no dup)", async () => { +test("APIKEY_PROVIDERS merges the 6 family files into 232 entries (no loss / no dup)", async () => { const keys = Object.keys((P as Record).APIKEY_PROVIDERS); - assert.equal(keys.length, 231); - assert.equal(new Set(keys).size, 231, "duplicate keys after spread-merge"); + assert.equal(keys.length, 232); + assert.equal(new Set(keys).size, 232, "duplicate keys after spread-merge"); // the merged object's entry-count equals the sum of the 6 semantic family files; families are a - // strict partition (every provider in exactly one), so the sum must be exactly 231. + // strict partition (every provider in exactly one), so the sum must be exactly 232. const families: [string, string][] = [ ["gateways", "APIKEY_PROVIDERS_GATEWAYS"], ["frontier-labs", "APIKEY_PROVIDERS_FRONTIER"], @@ -79,7 +79,7 @@ test("APIKEY_PROVIDERS merges the 6 family files into 231 entries (no loss / no seen.add(k); } } - assert.equal(famTotal, 231, "families must partition all 231 providers"); + assert.equal(famTotal, 232, "families must partition all 232 providers"); }); test("AI_PROVIDERS Proxy aggregates all sections; lookups resolve", () => { diff --git a/tests/unit/providers-patch-400.test.ts b/tests/unit/providers-patch-400.test.ts new file mode 100644 index 000000000000..8ec1cbdf3bc8 --- /dev/null +++ b/tests/unit/providers-patch-400.test.ts @@ -0,0 +1,42 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { updateProviderConnectionSchema } from "@/shared/validation/schemas/provider"; + +test("PATCH rateLimitOverrides {rpm:\"60\"} coerces to a valid number", () => { + const r = updateProviderConnectionSchema.safeParse({ rateLimitOverrides: { rpm: "60" } }); + assert.equal(r.success, true); +}); + +test("unknown key in rateLimitOverrides is rejected (no silent drop)", () => { + const r = updateProviderConnectionSchema.safeParse({ rateLimitOverrides: { rpm: 10, foo: 1 } }); + assert.equal(r.success, false); + const flaggedFoo = r.error!.issues.some( + (i) => i.path.includes("foo") || (i as { keys?: string[] }).keys?.includes("foo") || i.message.includes("foo") + ); + assert.ok( + flaggedFoo, + `expected an issue flagging "foo", got: ${JSON.stringify(r.error!.issues)}` + ); +}); + +test("quotaWindowThresholds key longer than 64 chars is rejected", () => { + const r = updateProviderConnectionSchema.safeParse({ + quotaWindowThresholds: { ["a".repeat(65)]: 50 }, + }); + assert.equal(r.success, false); +}); + +test("empty string rate limit value is rejected (coerce \"\"→0 trap)", () => { + const r = updateProviderConnectionSchema.safeParse({ rateLimitOverrides: { rpm: "" } }); + assert.equal(r.success, false); +}); + +test("non-numeric rate limit value is rejected", () => { + const r = updateProviderConnectionSchema.safeParse({ rateLimitOverrides: { rpm: "60abc" } }); + assert.equal(r.success, false); +}); + +test("quotaWindowThresholds value outside 0-100 is rejected", () => { + const r = updateProviderConnectionSchema.safeParse({ quotaWindowThresholds: { win: 101 } }); + assert.equal(r.success, false); +}); diff --git a/tests/unit/providers/uncloseai-noauth.test.ts b/tests/unit/providers-uncloseai-noauth.test.ts similarity index 100% rename from tests/unit/providers/uncloseai-noauth.test.ts rename to tests/unit/providers-uncloseai-noauth.test.ts diff --git a/tests/unit/quota-redis-store.test.ts b/tests/unit/quota-redis-store.test.ts index 3b9d5ded4c03..5a5ae936dc88 100644 --- a/tests/unit/quota-redis-store.test.ts +++ b/tests/unit/quota-redis-store.test.ts @@ -20,6 +20,9 @@ import assert from "node:assert/strict"; import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-redis-store-")); process.env.DATA_DIR = TEST_DATA_DIR; @@ -282,3 +285,16 @@ test("redisQuotaStore: resetRedisQuotaStore resets the store singleton", async ( // After reset, a new instance is created assert.ok(store2, "Should create new instance after reset"); }); + +test("redis namespace prefix: quota store KEY_PREFIX derives from REDIS_KEY_PREFIX", () => { + const quotaSrc = fs.readFileSync( + path.resolve(__dirname, "../../src/lib/quota/redisQuotaStore.ts"), + "utf8" + ); + assert.ok( + quotaSrc.includes('process.env.REDIS_KEY_PREFIX?.trim() || "omniroute:"') && + quotaSrc.includes("const KEY_PREFIX = `") && + quotaSrc.includes("quota`"), + "redisQuotaStore KEY_PREFIX must derive from REDIS_KEY_PREFIX env, defaulting to omniroute:quota" + ); +}); diff --git a/tests/unit/rate-limiter-redis-optional.test.ts b/tests/unit/rate-limiter-redis-optional.test.ts index edbce9b53d7d..8f23f3eafeee 100644 --- a/tests/unit/rate-limiter-redis-optional.test.ts +++ b/tests/unit/rate-limiter-redis-optional.test.ts @@ -41,3 +41,14 @@ test("#2357 checkRateLimit falls back when REDIS_URL is unset", () => { "checkRateLimit must route to the in-memory fallback when Redis is disabled" ); }); + +test("redis namespace prefix: rate limiter + auth cache keys are namespaced", () => { + assert.ok( + src.includes('process.env.REDIS_KEY_PREFIX?.trim() || "omniroute:"'), + "rateLimiter must read REDIS_KEY_PREFIX with an omniroute: default" + ); + assert.ok( + src.includes("keyPrefix: REDIS_KEY_PREFIX"), + "rateLimiter must pass the prefix as the ioredis keyPrefix so all keys are namespaced" + ); +}); diff --git a/tests/unit/reasoning-cache.test.ts b/tests/unit/reasoning-cache.test.ts index e13dfec7a95e..db538a803140 100644 --- a/tests/unit/reasoning-cache.test.ts +++ b/tests/unit/reasoning-cache.test.ts @@ -664,6 +664,7 @@ describe("Reasoning Replay Cache — Translator Replay", () => { { type: "reasoning", content: [{ type: "reasoning_text", text: "Cached Chat continuation reasoning" }], + summary: [], } ); }); @@ -751,47 +752,70 @@ describe("Reasoning Replay Cache — Translator Replay", () => { assert.equal(lookupReasoning(callId), "Authentic provider reasoning"); }); - it("should never cache Responses summaries or opaque plaintext companions", () => { - for (const [suffix, reasoningItem] of [ - [ - "summary", - { - type: "reasoning", - summary: [{ type: "summary_text", text: "Display-only summary" }], - }, - ], - [ - "mixed", - { - type: "reasoning", - encrypted_content: "opaque-provider-state", - content: [{ type: "reasoning_text", text: "Unsafe plaintext companion" }], - summary: [{ type: "summary_text", text: "Display-only mixed summary" }], - }, - ], - ] as const) { - clearReasoningCacheAll(); - const callId = `call_nonstream_${suffix}_reasoning`; - const translated = translateNonStreamingResponse( - { - object: "response", - model: "deepseek-v4-flash", - output: [ - reasoningItem, - { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, - ], - }, - FORMATS.OPENAI_RESPONSES, - FORMATS.OPENAI - ) as { choices?: Array<{ message?: Record }> }; - const message = translated.choices?.[0]?.message; - - assert.ok(message); - assert.equal(message.reasoning_content, undefined); - assert.ok(Array.isArray(message.reasoning_summary)); - assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 0); - assert.equal(lookupReasoning(callId), null); - } + it("preserves plaintext reasoning from a mixed plaintext + encrypted_content item (#10949)", () => { + clearReasoningCacheAll(); + const callId = "call_nonstream_mixed_reasoning"; + const translated = translateNonStreamingResponse( + { + object: "response", + model: "deepseek-v4-flash", + output: [ + { + type: "reasoning", + content: [ + { + type: "reasoning_text", + text: "Let me start by reading the directory to understand the structure of the corpus.", + }, + ], + encrypted_content: "", + summary: [], + }, + { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, + ], + }, + FORMATS.OPENAI_RESPONSES, + FORMATS.OPENAI + ) as { choices?: Array<{ message?: Record }> }; + const message = translated.choices?.[0]?.message; + + assert.ok(message); + assert.equal( + message.reasoning_content, + "Let me start by reading the directory to understand the structure of the corpus." + ); + assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 1); + assert.equal( + lookupReasoning(callId), + "Let me start by reading the directory to understand the structure of the corpus." + ); + }); + + it("should never cache summary-only Responses reasoning", () => { + clearReasoningCacheAll(); + const callId = "call_nonstream_summary_reasoning"; + const translated = translateNonStreamingResponse( + { + object: "response", + model: "deepseek-v4-flash", + output: [ + { + type: "reasoning", + summary: [{ type: "summary_text", text: "Display-only summary" }], + }, + { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, + ], + }, + FORMATS.OPENAI_RESPONSES, + FORMATS.OPENAI + ) as { choices?: Array<{ message?: Record }> }; + const message = translated.choices?.[0]?.message; + + assert.ok(message); + assert.equal(message.reasoning_content, undefined); + assert.ok(Array.isArray(message.reasoning_summary)); + assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 0); + assert.equal(lookupReasoning(callId), null); }); it("should preserve client-provided reasoning content", () => { diff --git a/tests/unit/reasoning-effort-clamp-and-retry.test.ts b/tests/unit/reasoning-effort-clamp-and-retry.test.ts new file mode 100644 index 000000000000..a97ac16df095 --- /dev/null +++ b/tests/unit/reasoning-effort-clamp-and-retry.test.ts @@ -0,0 +1,110 @@ +import { test, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { BaseExecutor } from "../../open-sse/executors/base.ts"; +import { + getLearnedReasoningEffort, + recordLearnedReasoningEffort, + __test_resetLearnedReasoningEffortCaps, +} from "../../open-sse/services/learnedReasoningEffortCaps.ts"; + +const OVH_422_BODY = JSON.stringify({ + error: { + message: + "Failed to deserialize the JSON body into the target type: reasoning_effort: " + + "unknown variant `xhigh`, expected one of `none`, `high`, `medium`, `low`, `minimal`", + }, +}); + +// Passthrough executor: returns the body unchanged so we assert on exactly what +// base.ts sends upstream. +class SimpleExecutor extends BaseExecutor { + constructor() { + super("openai-compatible-chat-eaff6869", { + baseUrls: ["https://oai.endpoints.kepler.ai.cloud.ovh.net/v1/chat/completions"], + }); + } + async transformRequest(_model: string, body: Record) { + return { ...body }; + } +} + +beforeEach(() => { + __test_resetLearnedReasoningEffortCaps(); +}); + +after(() => { + __test_resetLearnedReasoningEffortCaps(); +}); + +test("422 'unknown variant xhigh, expected one of ...' clamps reasoning_effort and retries once", async () => { + const executor = new SimpleExecutor(); + const originalFetch = globalThis.fetch; + const capturedBodies: Record[] = []; + + globalThis.fetch = async (_url: string | URL | Request, init: RequestInit = {}) => { + const body = JSON.parse(String(init.body)); + capturedBodies.push(body); + if (capturedBodies.length === 1) { + return new Response(OVH_422_BODY, { + status: 422, + headers: { "Content-Type": "application/json" }, + }); + } + return new Response(JSON.stringify({ ok: true }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + try { + const result = await executor.execute({ + model: "qwen3-coder-30b-a3b-instruct", + body: { reasoning_effort: "xhigh" }, + stream: false, + credentials: {}, + }); + assert.equal(capturedBodies.length, 2); + assert.equal(capturedBodies[0].reasoning_effort, "xhigh"); + assert.equal(capturedBodies[1].reasoning_effort, "high"); + assert.equal( + getLearnedReasoningEffort("openai-compatible-chat-eaff6869", "qwen3-coder-30b-a3b-instruct"), + "high" + ); + assert.equal(result.response.status, 200); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("a second request for the same provider+model sends the learned value on the first try", async () => { + const executor = new SimpleExecutor(); + const originalFetch = globalThis.fetch; + const capturedBodies: Record[] = []; + + globalThis.fetch = async (_url: string | URL | Request, init: RequestInit = {}) => { + const body = JSON.parse(String(init.body)); + capturedBodies.push(body); + return new Response(JSON.stringify({ ok: true }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + try { + recordLearnedReasoningEffort( + "openai-compatible-chat-eaff6869", + "qwen3-coder-30b-a3b-instruct", + ["none", "high", "medium", "low", "minimal"] + ); + await executor.execute({ + model: "qwen3-coder-30b-a3b-instruct", + body: { reasoning_effort: "xhigh" }, + stream: false, + credentials: {}, + }); + assert.equal(capturedBodies.length, 1); + assert.equal(capturedBodies[0].reasoning_effort, "high"); + } finally { + globalThis.fetch = originalFetch; + } +}); diff --git a/tests/unit/reasoning-effort-learned-capability.test.ts b/tests/unit/reasoning-effort-learned-capability.test.ts new file mode 100644 index 000000000000..210a451341e5 --- /dev/null +++ b/tests/unit/reasoning-effort-learned-capability.test.ts @@ -0,0 +1,91 @@ +import { test, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { sanitizeReasoningEffortForProvider } from "../../open-sse/executors/base/reasoningEffort.ts"; +import { + recordLearnedReasoningEffort, + __test_resetLearnedReasoningEffortCaps, +} from "../../open-sse/services/learnedReasoningEffortCaps.ts"; + +beforeEach(() => { + __test_resetLearnedReasoningEffortCaps(); +}); + +after(() => { + __test_resetLearnedReasoningEffortCaps(); +}); + +test("unregistered/custom provider+model: no learned cap yet sends xhigh unchanged", () => { + const body = { reasoning_effort: "xhigh" }; + const result = sanitizeReasoningEffortForProvider( + body, + "openai-compatible-chat-eaff6869", + "qwen3-coder-30b-a3b-instruct" + ) as { reasoning_effort: string }; + assert.equal(result.reasoning_effort, "xhigh"); +}); + +test("unregistered/custom provider+model: a learned cap clamps xhigh down to it", () => { + recordLearnedReasoningEffort("openai-compatible-chat-eaff6869", "qwen3-coder-30b-a3b-instruct", [ + "none", + "high", + "medium", + "low", + "minimal", + ]); + const body = { reasoning_effort: "xhigh" }; + const result = sanitizeReasoningEffortForProvider( + body, + "openai-compatible-chat-eaff6869", + "qwen3-coder-30b-a3b-instruct" + ) as { reasoning_effort: string }; + assert.equal(result.reasoning_effort, "high"); +}); + +test("learned cap only clamps when the requested effort is above it", () => { + recordLearnedReasoningEffort("acme", "model-x", ["none", "low", "medium"]); + const body = { reasoning_effort: "low" }; + const result = sanitizeReasoningEffortForProvider(body, "acme", "model-x") as { + reasoning_effort: string; + }; + assert.equal(result.reasoning_effort, "low"); +}); + +test("registry says supportsXHighEffort:false (and no supportsMax path) with a learned cap below 'high': uses the learned cap, not the hardcoded 'high'", () => { + // claude-haiku-4-5 is registered with supportsXHighEffort:false + // (open-sse/config/providers/registry/claude/index.ts) and its family is + // excluded from supportsClaudeMaxEffort (CLAUDE_MAX_EFFORT_UNSUPPORTED_FAMILY_PATTERNS + // in providerModels.ts), so it reaches the hardcoded-"high" line today — + // a real registry-covered case. Teach a lower cap and confirm it wins. + recordLearnedReasoningEffort("claude", "claude-haiku-4-5-20251001", ["none", "low", "medium"]); + const body = { reasoning_effort: "xhigh" }; + const result = sanitizeReasoningEffortForProvider( + body, + "claude", + "claude-haiku-4-5-20251001" + ) as { + reasoning_effort: string; + }; + assert.equal(result.reasoning_effort, "medium"); +}); + +test("registry says supportsXHighEffort:false with no learned cap: falls back to hardcoded 'high' (unchanged behavior)", () => { + const body = { reasoning_effort: "xhigh" }; + const result = sanitizeReasoningEffortForProvider( + body, + "claude", + "claude-haiku-4-5-20251001" + ) as { + reasoning_effort: string; + }; + assert.equal(result.reasoning_effort, "high"); +}); + +test("deepseek's non-ordinal max<->xhigh translation is untouched by the learned-cap catch-all", () => { + recordLearnedReasoningEffort("deepseek", "deepseek-v4", ["none", "low"]); + const body = { reasoning_effort: "xhigh" }; + const result = sanitizeReasoningEffortForProvider(body, "deepseek", "deepseek-v4") as { + reasoning_effort: string; + }; + // deepseek's special case returns early — xhigh -> max, never reaches the catch-all. + assert.equal(result.reasoning_effort, "max"); +}); diff --git a/tests/unit/reasoning-input-policy-summary-11108.test.ts b/tests/unit/reasoning-input-policy-summary-11108.test.ts new file mode 100644 index 000000000000..4317225dfb2b --- /dev/null +++ b/tests/unit/reasoning-input-policy-summary-11108.test.ts @@ -0,0 +1,145 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; + +const { applyReasoningInputPolicy } = + await import("../../open-sse/services/reasoningInputPolicy.ts"); + +test("#11108 applyReasoningInputPolicy defaults summary on a kept opaque reasoning item", () => { + const body: Record = { + input: [ + { + type: "reasoning", + id: "rs_example", + encrypted_content: "opaque-blob", + }, + ], + }; + + applyReasoningInputPolicy(body, "responses", { + provider: "opencode", + preserveEncryptedReasoning: true, + }); + + const input = body.input as Record[]; + assert.equal(input.length, 1); + assert.deepEqual(input[0].summary, []); +}); + +test("#11108 applyReasoningInputPolicy preserves an existing summary on a kept reasoning item", () => { + const body: Record = { + input: [ + { + type: "reasoning", + id: "rs_example", + encrypted_content: "opaque-blob", + summary: [{ type: "summary_text", text: "Planning." }], + }, + ], + }; + + applyReasoningInputPolicy(body, "responses", { + provider: "opencode", + preserveEncryptedReasoning: true, + }); + + const input = body.input as Record[]; + assert.deepEqual(input[0].summary, [{ type: "summary_text", text: "Planning." }]); +}); + +test("#11108 applyReasoningInputPolicy defaults summary on an opaque item surviving incompatible-drop", () => { + // Mixed item (plaintext + opaque) on an opaque-only transport is incompatible; + // dropIncompatibleResponsesReasoning() strips the plaintext content but keeps + // the opaque item alive — it must still get a default `summary`. + const body: Record = { + input: [ + { + type: "reasoning", + id: "rs_mixed", + content: [{ type: "reasoning_text", text: "inspect first" }], + encrypted_content: "opaque-blob", + }, + ], + }; + + const result = applyReasoningInputPolicy(body, "responses", { + provider: "codex", + onIncompatibleReasoning: "drop", + }); + + assert.equal(result.incompatibleReasoning, false); + const input = body.input as Record[]; + assert.equal(input.length, 1); + assert.equal(input[0].content, undefined); + assert.equal(input[0].encrypted_content, "opaque-blob"); + assert.deepEqual(input[0].summary, []); +}); + +test("#11108 applyReasoningInputPolicy strips a non-string id on a kept opaque reasoning item", () => { + // Same gap class as the summary fix above: opencode/zen also omits `id` + // entirely (surfaced by the client as `id: null`) on opaque-only reasoning + // items instead of a `rs_...` string. Replaying that shape verbatim trips + // strict Responses-API validators with "Expected 'id' to be a string." + const body: Record = { + input: [ + { + type: "reasoning", + id: null, + encrypted_content: "opaque-blob", + }, + ], + }; + + applyReasoningInputPolicy(body, "responses", { + provider: "opencode", + preserveEncryptedReasoning: true, + }); + + const input = body.input as Record[]; + assert.equal(input.length, 1); + assert.equal("id" in input[0], false); +}); + +test("#11108 applyReasoningInputPolicy strips a non-string id on a non-reasoning item (function_call)", () => { + // Same gap class, generic branch: any non-"reasoning" input item (function_call, + // message, ...) only stripped `id` when it was already a valid string, so a + // malformed `id` (e.g. `null`, mirroring the opencode/zen omission pattern) + // on a function_call item survived replay untouched. + const body: Record = { + input: [ + { + type: "function_call", + id: null, + call_id: "call_abc", + name: "bash", + arguments: "{}", + }, + ], + }; + + applyReasoningInputPolicy(body, "responses", { provider: "opencode" }); + + const input = body.input as Record[]; + assert.equal(input.length, 1); + assert.equal("id" in input[0], false); + assert.equal(input[0].call_id, "call_abc"); +}); + +test("#11108 applyReasoningInputPolicy preserves a valid string id on a kept opaque reasoning item", () => { + const body: Record = { + input: [ + { + type: "reasoning", + id: "rs_example", + encrypted_content: "opaque-blob", + }, + ], + }; + + applyReasoningInputPolicy(body, "responses", { + provider: "opencode", + preserveEncryptedReasoning: true, + }); + + const input = body.input as Record[]; + assert.equal(input[0].id, "rs_example"); +}); diff --git a/tests/unit/remove-hackclub-11118.test.ts b/tests/unit/remove-hackclub-11118.test.ts new file mode 100644 index 000000000000..9dc39fe1b32d --- /dev/null +++ b/tests/unit/remove-hackclub-11118.test.ts @@ -0,0 +1,7 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { REGISTRY } from "../../open-sse/config/providers/index.ts"; + +test("hackclub provider is removed from REGISTRY", () => { + assert.equal("hackclub" in REGISTRY, false); +}); diff --git a/tests/unit/responses-input-sanitizer-name.test.ts b/tests/unit/responses-input-sanitizer-name.test.ts index 7ac09604c7bc..4eec2d6c10e0 100644 --- a/tests/unit/responses-input-sanitizer-name.test.ts +++ b/tests/unit/responses-input-sanitizer-name.test.ts @@ -73,6 +73,23 @@ test("keeps valid server reasoning item ids", () => { assert.equal(result[0].id, "rs_123"); }); +test("strips a non-string reasoning item id instead of passing it through (#11108)", () => { + // Same gap class fixed in reasoningInputPolicy.ts: some upstreams (e.g. + // opencode/zen) send `id: null` instead of omitting it. The previous + // `typeof record.id !== "string"` guard returned the record unchanged in + // that case, letting a malformed id reach a strict Responses-API upstream. + const items = [ + { + id: null, + type: "reasoning", + summary: [{ type: "summary_text", text: "cached reasoning" }], + }, + ]; + const result = sanitizeResponsesInputItems(items) as Array>; + assert.equal("id" in result[0], false); + assert.equal(result[0].type, "reasoning"); +}); + test("normalizes user image_url content parts to input_image", () => { const items = [ { diff --git a/tests/unit/responses-parallel-tool-calls-index.test.ts b/tests/unit/responses-parallel-tool-calls-index.test.ts new file mode 100644 index 000000000000..f825f8700bac --- /dev/null +++ b/tests/unit/responses-parallel-tool-calls-index.test.ts @@ -0,0 +1,294 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const { openaiResponsesToOpenAIResponse } = + await import("../../open-sse/translator/response/openai-responses.ts"); + +// Issue: 2+ `function_call` items opened (response.output_item.added) before any of +// them closes (response.output_item.done) — a genuine parallel tool-call dispatch — +// causes `state.toolCallIndex` (only incremented in the `.done` handler) to stay at 0 +// for every "added" header chunk. Clients that key their tool-call accumulator by +// `delta.tool_calls[].index` (e.g. opencode's github-copilot chat-language-model +// stream parser) then see the *first* `.done` argument chunk at index 1/2 with no +// prior header and no `id`, and throw "Expected 'id' to be a string." +test("Responses -> OpenAI: parallel function_call items get distinct index+id on the added header", () => { + const state = {}; + + const added0 = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_0", name: "task" }, + }, + state + ); + const added1 = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_1", name: "task" }, + }, + state + ); + const added2 = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_2", name: "task" }, + }, + state + ); + + const headers = [added0, added1, added2].map((r) => r.choices[0].delta.tool_calls[0]); + + assert.deepEqual( + headers.map((h) => h.index), + [0, 1, 2], + "each parallel tool call must get its own header index, not all 0" + ); + assert.deepEqual( + headers.map((h) => h.id), + ["call_0", "call_1", "call_2"] + ); + + const done0 = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_0", name: "task", arguments: '{"i":0}' }, + }, + state + ); + const done1 = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_1", name: "task", arguments: '{"i":1}' }, + }, + state + ); + const done2 = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_2", name: "task", arguments: '{"i":2}' }, + }, + state + ); + + assert.deepEqual( + [done0, done1, done2].map((r) => r.choices[0].delta.tool_calls[0].index), + [0, 1, 2], + "argument chunks must reuse the SAME index assigned at .added time for each call_id" + ); +}); + +test("Responses -> OpenAI: parallel calls closed out of order keep their own index", () => { + const state = {}; + + for (const callId of ["call_a", "call_b", "call_c"]) { + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: callId, name: "task" }, + }, + state + ); + } + + // Close in reverse order: c, then a, then b. + const doneC = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_c", name: "task", arguments: "{}" }, + }, + state + ); + const doneA = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_a", name: "task", arguments: "{}" }, + }, + state + ); + const doneB = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_b", name: "task", arguments: "{}" }, + }, + state + ); + + assert.equal(doneC.choices[0].delta.tool_calls[0].index, 2); + assert.equal(doneA.choices[0].delta.tool_calls[0].index, 0); + assert.equal(doneB.choices[0].delta.tool_calls[0].index, 1); +}); + +test("Responses -> OpenAI: argument deltas interleaved across 2 parallel calls do not get glued together", () => { + const state = {}; + + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_x", name: "Read", id: "fc_call_x" }, + }, + state + ); + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_y", name: "Read", id: "fc_call_y" }, + }, + state + ); + + // Interleave argument deltas by item_id — x, y, x, y — before either closes. + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", item_id: "fc_call_x", delta: '{"filePath"' }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", item_id: "fc_call_y", delta: '{"filePath"' }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", item_id: "fc_call_x", delta: ':"/a.txt"}' }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", item_id: "fc_call_y", delta: ':"/b.txt"}' }, + state + ); + + const doneX = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_x", name: "Read" }, + }, + state + ); + const doneY = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_y", name: "Read" }, + }, + state + ); + + assert.equal(doneX.choices[0].delta.tool_calls[0].function.arguments, '{"filePath":"/a.txt"}'); + assert.equal(doneY.choices[0].delta.tool_calls[0].function.arguments, '{"filePath":"/b.txt"}'); +}); + +test("Responses -> OpenAI: a deferred (nameless) call that never resolves a name never consumes an index", () => { + const state = {}; + + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_deferred", name: "" }, + }, + state + ); + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_deferred", name: " " }, + }, + state + ); + + assert.equal(done, null); + assert.equal(state.toolCallIndex, 0); +}); + +test("Responses -> OpenAI: argument deltas interleaved across 2 parallel calls resolve by output_index when the upstream omits item_id", () => { + const state = {}; + + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + output_index: 0, + item: { type: "function_call", call_id: "call_p", name: "Read" }, + }, + state + ); + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + output_index: 1, + item: { type: "function_call", call_id: "call_q", name: "Read" }, + }, + state + ); + + // No item_id on any of these deltas — only output_index, which the Responses API + // guarantees on every streamed event regardless of whether item_id is also sent. + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", output_index: 0, delta: '{"filePath"' }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", output_index: 1, delta: '{"filePath"' }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", output_index: 0, delta: ':"/p.txt"}' }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", output_index: 1, delta: ':"/q.txt"}' }, + state + ); + + const doneP = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_p", name: "Read" }, + }, + state + ); + const doneQ = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "call_q", name: "Read" }, + }, + state + ); + + assert.equal(doneP.choices[0].delta.tool_calls[0].function.arguments, '{"filePath":"/p.txt"}'); + assert.equal(doneQ.choices[0].delta.tool_calls[0].function.arguments, '{"filePath":"/q.txt"}'); +}); + +test("Responses -> OpenAI: 2 parallel Agent calls still open at stream end each get their own flush chunk", () => { + const state = {}; + + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_agent0", name: "Agent" }, + }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", output_index: 0, delta: '{"task":"a"}' }, + state + ); + openaiResponsesToOpenAIResponse( + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "call_agent1", name: "Agent" }, + }, + state + ); + openaiResponsesToOpenAIResponse( + { type: "response.function_call_arguments.delta", output_index: 1, delta: '{"task":"b"}' }, + state + ); + + // Stream ends (chunk === null) before either call's output_item.done arrives. + const flushed = openaiResponsesToOpenAIResponse(null, state); + + assert.ok(Array.isArray(flushed)); + const argChunks = flushed.filter((c) => c.choices[0].delta.tool_calls); + assert.deepEqual( + argChunks.map((c) => c.choices[0].delta.tool_calls[0].index).sort(), + [0, 1], + "each still-open parallel call must get its own flush chunk, at its own index" + ); + const finalChunk = flushed[flushed.length - 1]; + assert.equal(finalChunk.choices[0].finish_reason, "tool_calls"); +}); diff --git a/tests/unit/search-blocked-providers-11100.test.ts b/tests/unit/search-blocked-providers-11100.test.ts new file mode 100644 index 000000000000..1c5c775e86ec --- /dev/null +++ b/tests/unit/search-blocked-providers-11100.test.ts @@ -0,0 +1,11 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { getAllSearchProviders } from "../../open-sse/config/searchRegistry.ts"; + +test("getAllSearchProviders filters out blocked providers", () => { + const all = getAllSearchProviders(); + assert.ok(all.some((p) => p.id === "serper-search")); + + const filtered = getAllSearchProviders(["serper-search"]); + assert.equal(filtered.some((p) => p.id === "serper-search"), false); +}); diff --git a/tests/unit/stream-continuation-wiring.test.ts b/tests/unit/stream-continuation-wiring.test.ts index 2f246e325d4b..36a6f445d1fc 100644 --- a/tests/unit/stream-continuation-wiring.test.ts +++ b/tests/unit/stream-continuation-wiring.test.ts @@ -47,9 +47,15 @@ async function collectText(stream: ReadableStream): Promise const ROLE = 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n'; const content = (s: string) => `data: {"choices":[{"delta":{"content":${JSON.stringify(s)}}}]}\n\n`; +const reasoning = (s: string) => + `data: {"choices":[{"delta":{"reasoning_content":${JSON.stringify(s)}}}]}\n\n`; +const finishStopNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'; +const finishLengthNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"length"}]}\n\n'; + test("mid-stream continuation: stitches the suffix after a silent post-commit truncation", async () => { - // Commits on chunk 1, emits "Hello wor", then ends WITHOUT a terminal marker (silent cut). - const initial = streamFrom([ROLE, content("Hello wor")]); + // Commits on chunk 1, emits "Hello there world", then ends WITHOUT a terminal marker + // (silent cut). + const initial = streamFrom([ROLE, content("Hello there world")]); let finalizeCount = 0; let continueArg = ""; @@ -60,15 +66,22 @@ test("mid-stream continuation: stitches the suffix after a silent post-commit tr now: steppingClock(), continueStream: async (soFar: string) => { continueArg = soFar; - // The model re-emits a small overlap ("wor") which must be trimmed away. - return streamFrom([ROLE, content("world!"), "data: [DONE]\n\n"]); + // The model re-emits only a partial tail of what was already sent ("there world", + // 11 chars — above the 8-char threshold, but NOT the full emitted text, unlike a + // full-string overlap this stays a discriminating test of trimContinuationOverlap's + // partial-tail trim, not just its "accept everything" path) before continuing. + return streamFrom([ROLE, content("there world, nice to meet you!"), "data: [DONE]\n\n"]); }, }); const out = await collectText(stream); const scan = scanOpenAiSseText(out); - assert.equal(continueArg, "Hello wor", "continuation is prefilled with the text already sent"); - assert.equal(scan.text, "Hello world!", "client sees the full answer, overlap trimmed, exactly once"); + assert.equal(continueArg, "Hello there world", "continuation is prefilled with the text already sent"); + assert.equal( + scan.text, + "Hello there world, nice to meet you!", + "client sees the full answer, partial overlap trimmed, exactly once" + ); assert.equal(scan.terminal, true, "the recovered stream ends with a terminal marker"); assert.equal(finalizeCount, 1, "finalize runs exactly once"); }); @@ -80,7 +93,7 @@ test("mid-stream continuation: recovers a post-commit transport error too", asyn const stream = createRecoverableStream(initial, async () => null, { finalize: () => {}, now: steppingClock(), - continueStream: async () => streamFrom([content("answer done."), "data: [DONE]\n\n"]), + continueStream: async () => streamFrom([content("Partial answer done."), "data: [DONE]\n\n"]), }); const scan = scanOpenAiSseText(await collectText(stream)); assert.equal(scan.text, "Partial answer done."); @@ -115,3 +128,167 @@ test("tool-call in flight is never continued (would corrupt tool JSON)", async ( await collectText(stream); assert.equal(continued, false, "continuation must NOT fire once a tool call has started streaming"); }); + +test("mid-stream continuation: a zero-overlap restart is rejected, never concatenated raw", async () => { + // Truncates silently after real, non-empty text — canContinue() fires. + const initial = streamFrom([ROLE, content("Tous les faits sont reunis")]); + let continuations = 0; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + maxContinuations: 1, + continueStream: async () => { + continuations += 1; + // The model ignores the assistant prefill and restarts on an unrelated sentence — + // zero characters of overlap with what was already emitted. + return streamFrom([ + content("Je complete le design - derniere verification"), + "data: [DONE]\n\n", + ]); + }, + }); + const out = await collectText(stream); + const scan = scanOpenAiSseText(out); + assert.equal( + scan.text, + "Tous les faits sont reunis", + "the unrelated restart must never be appended to the already-emitted text" + ); + assert.equal(scan.terminal, true, "closes cleanly instead of leaving the client hanging"); + assert.equal(continuations, 1, "bounded by maxContinuations — does not loop forever"); +}); + +test("mid-stream continuation: a nonzero overlap below the threshold is rejected too", async () => { + // Genuine 4-character overlap ("pret"), well under the 8-char threshold — this is the + // false-negative case a naive `overlapChars === 0` check would miss (a restart that + // happens to share a short accidental fragment with the emitted tail): must still be + // treated as a suspected restart, not accepted as a genuine resume. + const initial = streamFrom([ROLE, content("Le design est pret")]); + let continuations = 0; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + maxContinuations: 1, + continueStream: async () => { + continuations += 1; + // Shares only "pret" (4 chars) with the emitted tail, then diverges completely. + return streamFrom([content("pret a partir de zero"), "data: [DONE]\n\n"]); + }, + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal( + scan.text, + "Le design est pret", + "a below-threshold (but nonzero) overlap must not be accepted as a real resume" + ); + assert.equal(continuations, 1); +}); + +test("mid-stream continuation: a real overlap at or above the threshold is still stitched correctly", async () => { + // Regression guard: the existing happy path (first test in this file, whose updated + // fixture re-emits the 11-char partial tail "there world") still passes below — this test + // adds an overlap AT the threshold boundary to prove Task 3's new check does not fire when + // it shouldn't. + const initial = streamFrom([ROLE, content("The answer to this question")]); + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => + // "question" (8 chars) overlaps the tail of emittedText exactly at the threshold. + streamFrom([content("question is forty-two."), "data: [DONE]\n\n"]), + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal( + scan.text, + "The answer to this question is forty-two.", + "an overlap meeting the threshold is trimmed and stitched, not rejected" + ); +}); + +test("mid-stream continuation: a clean stop with reasoning-only output (no answer) triggers a continuation", async () => { + const initial = streamFrom([ + ROLE, + reasoning("the model thinks through the problem here..."), + finishStopNoContent, + ]); + let continueArg = "__unset__"; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async (soFar: string) => { + continueArg = soFar; + return streamFrom([content("Here is the actual answer."), "data: [DONE]\n\n"]); + }, + }); + const out = await collectText(stream); + const scan = scanOpenAiSseText(out); + assert.equal(continueArg, "", "nothing usable was emitted — the re-request has an empty prefill"); + assert.equal( + scan.text, + "Here is the actual answer.", + "the client gets a real answer instead of silence" + ); + assert.equal(scan.terminal, true); +}); + +test("mid-stream continuation: a clean stop with truly empty output (no text, no reasoning) is left unchanged", async () => { + const initial = streamFrom([ROLE, finishStopNoContent]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + await collectText(stream); + assert.equal( + continued, + false, + "no reasoning trace means there is nothing to act on — do not guess" + ); +}); + +test("mid-stream continuation: finish_reason 'length' with reasoning-only output does NOT trigger a continuation", async () => { + // Regression guard for a blocker found in cross-review: widening the gate to any + // terminal marker (instead of the literal finish_reason "stop") would wrongly spend a + // continuation attempt on a token-limit cutoff, which is out of this fix's scope. + const initial = streamFrom([ + ROLE, + reasoning("the model was still thinking when it hit the token limit..."), + finishLengthNoContent, + ]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + await collectText(stream); + assert.equal(continued, false, "finish_reason 'length' is out of scope for this fix"); +}); + +test("mid-stream continuation: real content alongside reasoning at a clean stop is left unchanged (non-regression)", async () => { + const initial = streamFrom([ + ROLE, + reasoning("thinking..."), + content("The real answer."), + finishStopNoContent, + ]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal(continued, false, "real content was delivered — nothing to recover"); + assert.equal(scan.text, "The real answer."); +}); diff --git a/tests/unit/stream-continuation.test.ts b/tests/unit/stream-continuation.test.ts index caffe936df9d..8a1da37ce9d6 100644 --- a/tests/unit/stream-continuation.test.ts +++ b/tests/unit/stream-continuation.test.ts @@ -21,6 +21,30 @@ test("scanOpenAiSseText accumulates content deltas and flags an OpenAI-compat st assert.equal(r.terminal, false); }); +test("scanOpenAiSseText accumulates reasoning_content deltas separately from content", () => { + const sse = + 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n' + + 'data: {"choices":[{"delta":{"reasoning_content":"thinking..."}}]}\n\n' + + 'data: {"choices":[{"delta":{"reasoning_content":" more"}}]}\n\n'; + const r = scanOpenAiSseText(sse); + assert.equal(r.reasoningText, "thinking... more"); + assert.equal(r.text, "", "reasoning_content must never leak into the visible text field"); + assert.equal(r.parsedOpenAi, true); +}); + +test("scanOpenAiSseText captures the literal finish_reason value", () => { + const stop = scanOpenAiSseText('data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'); + assert.equal(stop.finishReason, "stop"); + + const length = scanOpenAiSseText( + 'data: {"choices":[{"delta":{"content":"x"},"finish_reason":"length"}]}\n\n' + ); + assert.equal(length.finishReason, "length"); + + const none = scanOpenAiSseText('data: {"choices":[{"delta":{"content":"x"}}]}\n\n'); + assert.equal(none.finishReason, null, "no finish_reason seen means null, not a guessed default"); +}); + test("scanOpenAiSseText detects the terminal [DONE] marker", () => { const r = scanOpenAiSseText('data: {"choices":[{"delta":{"content":"hi"}}]}\n\ndata: [DONE]\n\n'); assert.equal(r.text, "hi"); @@ -65,6 +89,15 @@ test("makeContinuationBody refuses bodies without a messages array or empty text assert.equal(makeContinuationBody(null as never, "t"), null); }); +test("makeContinuationBody accepts an empty prefill by re-sending the messages unchanged", () => { + const body = { model: "x", stream: true, messages: [{ role: "user", content: "hi" }] }; + const out = makeContinuationBody(body, ""); + assert.ok(out, "an empty prefill must still produce a re-request body, not null"); + assert.equal(out!.messages.length, 1, "no empty assistant turn is appended"); + assert.deepEqual(out!.messages[0], { role: "user", content: "hi" }); + assert.equal(out!.stream, true); +}); + // ── trimContinuationOverlap ─────────────────────────────────────────────────── test("trimContinuationOverlap removes a duplicated seam so the join is append-only", () => { diff --git a/tests/unit/stream-handler.test.ts b/tests/unit/stream-handler.test.ts index d19a13f351e4..8c19c088021b 100644 --- a/tests/unit/stream-handler.test.ts +++ b/tests/unit/stream-handler.test.ts @@ -167,6 +167,41 @@ test("createDisconnectAwareStream treats cancel after Responses completed as suc assert.equal(disconnectHandled, false); }); +test("createDisconnectAwareStream recognizes a large Responses compaction completion", async () => { + let errorHandled = false; + const completed = `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + status: "completed", + output: [{ type: "compaction", encrypted_content: "x".repeat(5000) }], + }, + })}\n\n`; + const transformStream = { + readable: new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(completed)); + controller.close(); + }, + }), + writable: createNoopAbortWritable(), + }; + + const stream = createDisconnectAwareStream( + transformStream, + createStreamController({ + clientResponseFormat: FORMATS.OPENAI_RESPONSES, + onError() { + errorHandled = true; + }, + }) + ); + const text = await readStreamText(stream); + + assert.equal(text, completed); + assert.equal(errorHandled, false); + assert.doesNotMatch(text, /response\.failed/); +}); + test("createDisconnectAwareStream: Gemini 503 high-demand error becomes SSE error chunk with message preserved", async () => { const geminiMsg = "[503]: This model is currently experiencing high demand. Spikes in demand are usually temporary. Please try again later."; diff --git a/tests/unit/stream-payload-collector.test.ts b/tests/unit/stream-payload-collector.test.ts index e8869bb8bfe1..63b96c5eaf60 100644 --- a/tests/unit/stream-payload-collector.test.ts +++ b/tests/unit/stream-payload-collector.test.ts @@ -2,6 +2,7 @@ import test from "node:test"; import assert from "node:assert/strict"; const collector = await import("../../open-sse/utils/streamPayloadCollector.ts"); +import { splitConcatenatedToolCallArguments } from "../../open-sse/utils/streamPayloadCollector.ts"; test("compactStructuredStreamPayload returns null for null input", () => { assert.equal(collector.compactStructuredStreamPayload(null), null); @@ -413,3 +414,33 @@ test("#9315: getSummary() returns undefined when no format was configured (unaff c.push({ choices: [{ index: 0, delta: { content: "hi" } }] }); assert.equal(c.getSummary(), undefined); }); + +test("splitConcatenatedToolCallArguments — two back-to-back JSON objects", () => { + const a = JSON.stringify({ tool: "x", args: "1" }); + const b = JSON.stringify({ tool: "y", args: "2" }); + const out = splitConcatenatedToolCallArguments(a + b); + assert.deepEqual(out, [a, b]); // >=2 valid values -> split (array of parts) +}); + +test("splitConcatenatedToolCallArguments — nested object + escaped quotes stay single JSON", () => { + const a = JSON.stringify({ a: 'he said "hi"', b: { c: 1 } }); + const single = a; // a is ONE valid JSON object -> no split + const out = splitConcatenatedToolCallArguments(single); + assert.equal(out, null); // single valid JSON -> untouched (null) +}); + +test("splitConcatenatedToolCallArguments — braces/quotes inside strings exercise escaped scanner", () => { + // Two valid JSON values whose string bodies contain braces and escaped quotes. + // Concatenated they reach the inString/escaped state machine (not the JSON.parse + // fast path), so this covers the case the owner asked about. + const a = JSON.stringify({ cmd: 'echo "}{" ; x' }); + const b = JSON.stringify({ cmd: "{[not json]}" }); + const out = splitConcatenatedToolCallArguments(a + b); + assert.deepEqual(out, [a, b]); // >=2 valid values -> split into parts +}); + +test("splitConcatenatedToolCallArguments — top-level array is single value", () => { + const arr = JSON.stringify([{ tool: "x" }, { tool: "y" }]); + const out = splitConcatenatedToolCallArguments(arr); + assert.equal(out, null); // one value boundary (array) -> not split +}); diff --git a/tests/unit/stream-recovery-toolcall.test.ts b/tests/unit/stream-recovery-toolcall.test.ts new file mode 100644 index 000000000000..7f3a76534332 --- /dev/null +++ b/tests/unit/stream-recovery-toolcall.test.ts @@ -0,0 +1,178 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + createRecoverableStream, + TruncatedStreamError, + scanOpenAiSseText, +} from "../../open-sse/services/streamRecovery.ts"; + +const enc = new TextEncoder(); + +// Deliver the SSE chunk on the first read, then error on the second read so the +// holdback window has committed (post-commit truncation) before the cut. +function makeStream(sse: string): ReadableStream { + let n = 0; + return new ReadableStream({ + pull(c) { + n += 1; + if (n === 1) { + c.enqueue(enc.encode(sse)); + return; + } + c.error(new TruncatedStreamError()); + }, + }); +} + +// A clock that jumps past HOLDBACK_MS on the second read so the very first pushed +// chunk commits the holdback window immediately (post-commit truncation path). +function jumpingClock(): () => number { + let t = 0; + return () => (t += 1000); +} + +describe("scanOpenAiSseText: terminal vs in-flight tool call", () => { + it("tool_calls without finish_reason → inFlight true, terminal false", () => { + const sse = + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"call_1","function":{"name":"lookup"}}]}}]}\n\n'; + const r = scanOpenAiSseText(sse); + assert.equal(r.sawToolCall, true); + assert.equal(r.sawToolCallInFlight, true); + assert.equal(r.terminal, false); + }); + + it("complete tool_calls + finish_reason + [DONE] → terminal true, inFlight false", () => { + const sse = + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"call_1","function":{"name":"lookup","arguments":"{}"}}]}}]}\n' + + 'data: {"choices":[{"index":0,"delta":{},"finish_reason":"tool_calls"}]}\n' + + "data: [DONE]\n\n"; + const r = scanOpenAiSseText(sse); + assert.equal(r.sawToolCall, true); + assert.equal(r.terminal, true); + assert.equal(r.sawToolCallInFlight, false); + }); + + it("plain text → no tool call", () => { + const sse = 'data: {"choices":[{"index":0,"delta":{"content":"hello"}}]}\n\n'; + const r = scanOpenAiSseText(sse); + assert.equal(r.sawToolCall, false); + assert.equal(r.sawToolCallInFlight, false); + assert.equal(r.terminal, false); + }); + + it("complete tool_calls WITHOUT [DONE] → terminal false, inFlight false (the actual fix)", () => { + // This is the case the original plan promised to unblock: the tool call itself is + // done (finish_reason: "tool_calls"), but the overall stream/turn has not sent its + // own terminal marker yet — a truncation right here is recoverable. + const sse = + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"call_1","function":{"name":"lookup","arguments":"{}"}}]}}]}\n' + + 'data: {"choices":[{"index":0,"delta":{},"finish_reason":"tool_calls"}]}\n\n'; + const r = scanOpenAiSseText(sse); + assert.equal(r.sawToolCall, true); + assert.equal(r.sawToolCallInFlight, false); + assert.equal(r.terminal, false); + }); +}); + +describe("stream recovery does not duplicate an in-flight tool call", () => { + it("truncation with an in-flight tool call → no continuation", async () => { + let continued = false; + const sse = + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"c1","function":{"name":"f"}}]}}]}\n\n'; + const wrapped = createRecoverableStream(makeStream(sse), async () => null, { + finalize: () => {}, + now: jumpingClock(), + continueStream: async () => { + continued = true; + return null; + }, + }); + const reader = wrapped.getReader(); + try { + for (;;) { + const r = await reader.read(); + if (r.done) break; + } + } catch { + // the in-flight tool call makes the stream close without continuing + } + assert.equal(continued, false); + }); + + it("truncation right after a completed tool call → continuation attempted (the real 91% gain)", async () => { + // Text was emitted, THEN the tool call completed (finish_reason: "tool_calls"), THEN + // the connection drops before a [DONE]/other terminal marker. Before this fix, the + // blunt `emittedToolCall` guard blocked recovery here even though the call itself is + // done and only trailing prose was lost — this is the exact case the plan promised + // to unblock and the pre-fix table proved was a no-op. + let continued = false; + const sse = + 'data: {"choices":[{"index":0,"delta":{"content":"Let me check that. "}}]}\n' + + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"c1","function":{"name":"f","arguments":"{}"}}]}}]}\n' + + 'data: {"choices":[{"index":0,"delta":{},"finish_reason":"tool_calls"}]}\n\n'; + const wrapped = createRecoverableStream(makeStream(sse), async () => null, { + finalize: () => {}, + now: jumpingClock(), + continueStream: async () => { + continued = true; + return null; + }, + }); + const reader = wrapped.getReader(); + try { + for (;;) { + const r = await reader.read(); + if (r.done) break; + } + } catch { + // no-op + } + assert.equal(continued, true); + }); + + it("truncation of plain text → continuation attempted", async () => { + let continued = false; + const sse = 'data: {"choices":[{"index":0,"delta":{"content":"hello "}}]}\n\n'; + const wrapped = createRecoverableStream(makeStream(sse), async () => null, { + finalize: () => {}, + now: jumpingClock(), + continueStream: async () => { + continued = true; + return null; + }, + }); + const reader = wrapped.getReader(); + try { + for (;;) { + const r = await reader.read(); + if (r.done) break; + } + } catch { + // no-op + } + assert.equal(continued, true); + }); + + it("naive removal of the tool-call guard would duplicate a partial tool call", () => { + // The blunt `sawToolCall` flag is true for BOTH a complete tool call and a + // partial (in-flight) one. The new `sawToolCallInFlight` flag is the only + // signal that tells them apart: a naive guard keyed on `sawToolCall` would + // block the complete call AND let the partial one through to the + // continuation, where trimContinuationOverlap (text-only) cannot de-duplicate + // the replayed tool_calls arguments. + const ssePartial = + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"call_1","function":{"name":"lookup","arguments":"{\\"q\\""}}]}}]}\n\n'; + const sseFull = + 'data: {"choices":[{"index":0,"delta":{"tool_calls":[{"id":"call_1","function":{"name":"lookup","arguments":"{\\"q\\":\\"x\\"}"}}]}}]}\n' + + 'data: {"choices":[{"index":0,"delta":{},"finish_reason":"tool_calls"}]}\n\n'; + const scanPartial = scanOpenAiSseText(ssePartial); + const scanFull = scanOpenAiSseText(sseFull); + // The blunt flag cannot distinguish them. + assert.equal(scanPartial.sawToolCall, true); + assert.equal(scanFull.sawToolCall, true); + // The in-flight flag can — and that is what keeps canContinue false only for + // the partial tool call, so the continuation never replays it. + assert.equal(scanPartial.sawToolCallInFlight, true); + assert.equal(scanFull.sawToolCallInFlight, false); + }); +}); diff --git a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts index cab5e9c37f53..8b3059236dd1 100644 --- a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts +++ b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts @@ -8,7 +8,7 @@ import { omitEncryptedReasoningForLog } from "../../src/lib/logPayloads.ts"; // Responses reasoning replay is target-scoped. Plaintext DeepSeek state and // provider-generated opaque state are never interchangeable. -test("unknown Responses targets reject opaque reasoning and ignore display summaries", () => { +test("unknown Responses targets drop opaque reasoning and preserve display summaries (#10959)", () => { const body: Record = { input: [ { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, @@ -24,11 +24,14 @@ test("unknown Responses targets reject opaque reasoning and ignore display summa ], }; - const originalInput = structuredClone(body.input); const result = applyReasoningInputPolicy(body, "responses"); - assert.equal(result.incompatibleReasoning, true); - assert.deepEqual(body.input, originalInput, "rejection must not mutate the request"); + assert.equal(result.incompatibleReasoning, false); + assert.deepEqual(body.input, [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "reasoning", summary: [{ text: "display only" }] }, + { type: "function_call", name: "search", arguments: "{}", call_id: "call_1" }, + ]); }); test("unannotated targets preserve plaintext Responses reasoning without synthetic IDs", () => { @@ -132,7 +135,7 @@ test("Chat drop removes opaque state while preserving plaintext and summary deta ]); }); -test("DeepSeek rejects plaintext reasoning carrying opaque provider state", () => { +test("DeepSeek projects plaintext reasoning carrying opaque provider state onto the plaintext transport (#10949)", () => { for (const opaqueField of ["signature", "format"] as const) { const body: Record = { input: [ @@ -148,13 +151,11 @@ test("DeepSeek rejects plaintext reasoning carrying opaque provider state", () = const result = applyReasoningInputPolicy(body, "responses", { provider: "deepseek" }); - assert.equal(result.incompatibleReasoning, true, opaqueField); + assert.equal(result.incompatibleReasoning, false, opaqueField); assert.deepEqual(body.input, [ { - id: "rs_mixed123", type: "reasoning", content: [{ type: "reasoning_text", text: "untrusted companion" }], - [opaqueField]: "provider-state", }, { type: "message", role: "user", content: [{ type: "input_text", text: "continue" }] }, ]); @@ -199,6 +200,49 @@ test("drop fallback removes only the incompatible active transport and preserves assert.equal(opaqueReasoning.encrypted_content, "provider-state"); }); +test("mixed plaintext + opaque reasoning follows the target transport instead of rejecting (#10949)", () => { + const mixedReasoning = { + id: "rs_mixed", + type: "reasoning", + content: [{ type: "reasoning_text", text: "inspect first" }], + encrypted_content: "provider-state", + summary: [{ type: "summary_text", text: "display only" }], + }; + + // Plaintext target (deepseek): keep the portable plaintext, strip opaque state. + const toPlaintext: Record = { + input: [structuredClone(mixedReasoning)], + }; + const plaintextResult = applyReasoningInputPolicy(toPlaintext, "responses", { + provider: "deepseek", + }); + assert.equal(plaintextResult.incompatibleReasoning, false); + assert.deepEqual(toPlaintext.input, [ + { + type: "reasoning", + content: [{ type: "reasoning_text", text: "inspect first" }], + summary: [{ type: "summary_text", text: "display only" }], + }, + ]); + + // Opaque target (openai): keep the provider state, strip the plaintext. + const toOpaque: Record = { + input: [structuredClone(mixedReasoning)], + }; + const opaqueResult = applyReasoningInputPolicy(toOpaque, "responses", { + provider: "openai", + }); + assert.equal(opaqueResult.incompatibleReasoning, false); + assert.deepEqual(toOpaque.input, [ + { + id: "rs_mixed", + type: "reasoning", + encrypted_content: "provider-state", + summary: [{ type: "summary_text", text: "display only" }], + }, + ]); +}); + test("drop fallback preserves reasoning when its transport is compatible", () => { const body: Record = { input: [ @@ -274,7 +318,11 @@ test("explicit custom target opt-in remains an opaque transport override", () => const result = applyReasoningInputPolicy(body, "responses", { preserveEncryptedReasoning: true }); assert.equal(result.incompatibleReasoning, false); - assert.deepEqual(body.input, [{ type: "reasoning", encrypted_content: "encrypted-blob" }]); + // #11108: a kept opaque item defaults `summary` when the source omitted it — + // some upstreams reject `input[]` reasoning items missing the field entirely. + assert.deepEqual(body.input, [ + { type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }, + ]); }); test("preserved opaque reasoning remains redacted from log copies", () => { diff --git a/tests/unit/token-health-check-kimi.test.ts b/tests/unit/token-health-check-kimi.test.ts index c5d487762603..87b816952879 100644 --- a/tests/unit/token-health-check-kimi.test.ts +++ b/tests/unit/token-health-check-kimi.test.ts @@ -18,11 +18,10 @@ describe("Kimi Background Health Sweep", () => { it("triggers refresh when Kimi token is within jittered expiration window", async () => { const nowSec = Math.floor(Date.now() / 1000); - // Token expiring in 30s. The production jitter threshold is 60-240s, so - // remainingSec=30 is always <= threshold and must trigger a refresh. + // Token expiring in 90 seconds (within 60-240s window) const token = "eyJhbGciOiJIUzUxMiJ9." + - Buffer.from(JSON.stringify({ exp: nowSec + 30, iat: nowSec })).toString("base64url") + + Buffer.from(JSON.stringify({ exp: nowSec + 90, iat: nowSec })).toString("base64url") + ".sig"; let calledRefresh = false; diff --git a/tests/unit/translator-openai-responses-req.test.ts b/tests/unit/translator-openai-responses-req.test.ts index e0d0a5177318..2a05f57310e8 100644 --- a/tests/unit/translator-openai-responses-req.test.ts +++ b/tests/unit/translator-openai-responses-req.test.ts @@ -172,28 +172,26 @@ test("Responses -> Chat keeps summary-only reasoning out of continuation state", assert.equal(result.messages[0].reasoning_content, undefined); }); -test("Responses -> Chat rejects opaque reasoning instead of replaying its plaintext companion", () => { - assert.throws( - () => - openaiResponsesToOpenAIRequest( - "deepseek-v4-pro", +test("Responses -> Chat replays the plaintext companion of an opaque reasoning item (#10949)", () => { + const result = openaiResponsesToOpenAIRequest( + "deepseek-v4-pro", + { + input: [ { - input: [ - { - id: "rs_opaque", - type: "reasoning", - encrypted_content: "opaque-provider-state", - content: [{ type: "reasoning_text", text: "Untrusted plaintext companion" }], - summary: [{ type: "summary_text", text: "Display summary" }], - }, - { type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, - ], + id: "rs_opaque", + type: "reasoning", + encrypted_content: "opaque-provider-state", + content: [{ type: "reasoning_text", text: "Untrusted plaintext companion" }], + summary: [{ type: "summary_text", text: "Display summary" }], }, - false, - { _preserveReasoningContent: true } - ), - /Reasoning continuation is not compatible/ - ); + { type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, + ], + }, + false, + { _preserveReasoningContent: true } + ) as { messages: Array> }; + + assert.equal(result.messages[0].reasoning_content, "Untrusted plaintext companion"); }); test("Responses -> Chat merges assistant text that follows a function call", () => { @@ -475,6 +473,7 @@ test("Chat -> Responses defaults unannotated targets to plaintext reasoning", () { type: "reasoning", content: [{ type: "reasoning_text", text: "Inspect the repository first" }], + summary: [], }, { type: "function_call", @@ -513,6 +512,7 @@ test("Chat -> DeepSeek Responses accepts the plaintext reasoning alias", () => { assert.deepEqual(result.input[0], { type: "reasoning", content: [{ type: "reasoning_text", text: "Alias plaintext reasoning" }], + summary: [], }); }); diff --git a/tests/unit/translator-resp-openai-responses.test.ts b/tests/unit/translator-resp-openai-responses.test.ts index 27cc877f2ef2..2253c655ac7a 100644 --- a/tests/unit/translator-resp-openai-responses.test.ts +++ b/tests/unit/translator-resp-openai-responses.test.ts @@ -423,6 +423,52 @@ test("Responses -> OpenAI: preserves non-object Read JSON-string arguments", () assert.equal(done.choices[0].delta.tool_calls[0].function.arguments, "null"); }); +test("Responses -> OpenAI: mixed plaintext + encrypted_content reasoning replays its plaintext (#10949)", () => { + const state = {}; + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { + type: "reasoning", + id: "rs_mixed", + content: [ + { + type: "reasoning_text", + text: "Let me start by reading the directory to understand the structure of the corpus.", + }, + ], + encrypted_content: "", + summary: [], + }, + }, + state + ); + + assert.ok(done, "mixed reasoning item must surface a delta"); + assert.equal( + done.choices[0].delta.reasoning_content, + "Let me start by reading the directory to understand the structure of the corpus." + ); +}); + +test("Responses -> OpenAI: opaque-only reasoning still emits no fabricated plaintext", () => { + const state = {}; + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { + type: "reasoning", + id: "rs_opaque_only", + encrypted_content: "", + summary: [], + }, + }, + state + ); + + assert.equal(done, null); +}); + test("Responses -> OpenAI: strips empty optional args from JSON-string output_item.done arguments", () => { const state = {}; openaiResponsesToOpenAIResponse( diff --git a/tests/unit/translator-resp-openai-to-claude.test.ts b/tests/unit/translator-resp-openai-to-claude.test.ts index 59e58c05220f..f8f1cfcfdbd6 100644 --- a/tests/unit/translator-resp-openai-to-claude.test.ts +++ b/tests/unit/translator-resp-openai-to-claude.test.ts @@ -105,7 +105,9 @@ test("OpenAI stream: internal reasoning replay placeholder stays hidden from Cla const result = flatten([placeholder, text]); assert.equal( - result.some((event) => event.type === "content_block_start" && event.content_block?.type === "thinking"), + result.some( + (event) => event.type === "content_block_start" && event.content_block?.type === "thinking" + ), false ); assert.equal(result[0].type, "message_start"); @@ -217,10 +219,7 @@ test("OpenAI stream: multi-chunk content without the placeholder passes through textDeltas.map((event) => event.delta.text), ["Hello, ", "world.", " Bye."] ); - assert.equal( - textDeltas.map((event) => event.delta.text).join(""), - "Hello, world. Bye." - ); + assert.equal(textDeltas.map((event) => event.delta.text).join(""), "Hello, world. Bye."); }); test("OpenAI stream: tool calls strip Claude OAuth prefix and keep cache usage", () => { @@ -427,9 +426,11 @@ test("OpenAI stream: XML block in content becomes tool_use at finish", // message_start → (no text block since all content was XML) assert.equal(result[0].type, "message_start"); // At finish: tool_use content_block_start - const toolStart = result.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + const toolStart = result.find( + (e) => e.type === "content_block_start" && e.content_block?.type === "tool_use" + ); assert.ok(toolStart, "expected tool_use content_block_start"); - assert.equal(toolStart.content_block.name, "bash"); // normalized via REVERSE_MAP + assert.equal(toolStart.content_block.name, "Bash"); // canonical echo kept (#11085 live repro) assert.deepEqual(toolStart.content_block.input, { command: "ls -la" }); // tool_use content_block_stop const toolStop = result.find((e) => e.type === "content_block_stop"); @@ -465,7 +466,7 @@ test("OpenAI stream: XML invoke block across two streaming chunks", () => { choices: [ { index: 0, - delta: { content: 'hosts' }, + delta: { content: "hosts" }, finish_reason: null, }, ], @@ -487,9 +488,11 @@ test("OpenAI stream: XML invoke block across two streaming chunks", () => { // Buffer should be cleared after chunk2 assert.equal(state._xmlInvokeBuffer, "", "buffer cleared after complete block"); - const toolStart = result.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + const toolStart = result.find( + (e) => e.type === "content_block_start" && e.content_block?.type === "tool_use" + ); assert.ok(toolStart, "expected tool_use content_block_start"); - assert.equal(toolStart.content_block.name, "read"); + assert.equal(toolStart.content_block.name, "Read"); // canonical echo kept (#11085 live repro) assert.deepEqual(toolStart.content_block.input, { file_path: "/etc/hosts" }); }); @@ -538,15 +541,25 @@ test("OpenAI stream: text before XML block is emitted as text content", () => { const result = flatten([chunk1, chunk2, chunk3]); // "Checking..." should be emitted as text - const textDeltas = result.filter((e) => e.type === "content_block_delta" && e.delta?.type === "text_delta"); + const textDeltas = result.filter( + (e) => e.type === "content_block_delta" && e.delta?.type === "text_delta" + ); assert.ok(textDeltas.length > 0, "expected at least one text delta"); - assert.ok(textDeltas.some((d) => d.delta.text.includes("Checking...")), "text before XML preserved"); - assert.ok(textDeltas.some((d) => d.delta.text.includes("Done.")), "text after XML preserved"); + assert.ok( + textDeltas.some((d) => d.delta.text.includes("Checking...")), + "text before XML preserved" + ); + assert.ok( + textDeltas.some((d) => d.delta.text.includes("Done.")), + "text after XML preserved" + ); // Tool call should still be emitted - const toolStart = result.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + const toolStart = result.find( + (e) => e.type === "content_block_start" && e.content_block?.type === "tool_use" + ); assert.ok(toolStart, "expected tool_use content_block_start"); - assert.equal(toolStart.content_block.name, "bash"); + assert.equal(toolStart.content_block.name, "Bash"); // canonical echo kept (#11085 live repro) assert.deepEqual(toolStart.content_block.input, { command: "date" }); }); diff --git a/tests/unit/ui/add-api-key-modal-enter-key.test.ts b/tests/unit/ui/add-api-key-modal-enter-key.test.ts deleted file mode 100644 index 2532eb297ea8..000000000000 --- a/tests/unit/ui/add-api-key-modal-enter-key.test.ts +++ /dev/null @@ -1,26 +0,0 @@ -import { describe, it } from "node:test"; -import assert from "node:assert/strict"; -import fs from "node:fs"; -import path from "node:path"; - -describe("AddApiKeyModal Enter key submit (#10995)", () => { - it("AddApiKeyModal attaches onKeyDown Enter handler to the API Key input", () => { - const modalPath = path.resolve( - process.cwd(), - "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx" - ); - const content = fs.readFileSync(modalPath, "utf8"); - assert.ok( - content.includes("onKeyDown"), - "AddApiKeyModal must contain onKeyDown event handler for Enter key validation" - ); - assert.ok( - content.includes('e.key === "Enter"'), - "onKeyDown handler must check for Enter key press" - ); - assert.ok( - content.includes("handleValidate()"), - "Enter key press must invoke handleValidate()" - ); - }); -}); diff --git a/tests/unit/ui/add-api-key-modal-enter-key.test.tsx b/tests/unit/ui/add-api-key-modal-enter-key.test.tsx new file mode 100644 index 000000000000..1c28a13230f0 --- /dev/null +++ b/tests/unit/ui/add-api-key-modal-enter-key.test.tsx @@ -0,0 +1,104 @@ +// @vitest-environment jsdom +// +// #10995 — Enter key in AddApiKeyModal triggers key validation without requiring a mouse click on Check. +import React, { act } from "react"; +import { createRoot } from "react-dom/client"; +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +vi.mock("next-intl", () => ({ + useTranslations: () => (key: string) => key, +})); + +const { default: AddApiKeyModal } = + await import("../../../src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal"); + +const containers: Array<{ root: ReturnType; el: HTMLDivElement }> = []; + +function render(props: Record) { + const el = document.createElement("div"); + document.body.appendChild(el); + const root = createRoot(el); + act(() => { + root.render( + undefined} + onClose={() => {}} + {...(props as any)} + /> + ); + }); + containers.push({ root, el }); + return el; +} + +function setInputValue(input: HTMLInputElement, value: string) { + const setter = Object.getOwnPropertyDescriptor(window.HTMLInputElement.prototype, "value")!.set!; + act(() => { + setter.call(input, value); + input.dispatchEvent(new Event("input", { bubbles: true })); + }); +} + +function dispatchKeyDown(element: HTMLElement, key: string) { + act(() => { + element.dispatchEvent(new KeyboardEvent("keydown", { key, bubbles: true, cancelable: true })); + }); +} + +describe("AddApiKeyModal Enter key submit (#10995)", () => { + let originalFetch: typeof global.fetch; + + beforeEach(() => { + originalFetch = global.fetch; + }); + + afterEach(() => { + global.fetch = originalFetch; + for (const { root, el } of containers) { + act(() => root.unmount()); + el.remove(); + } + containers.length = 0; + }); + + it("does not trigger validation on Enter when input is empty", () => { + const fetchMock = vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ valid: true }), + }); + global.fetch = fetchMock as any; + + const el = render({}); + const input = el.querySelector('input[type="password"]'); + expect(input).toBeTruthy(); + + dispatchKeyDown(input!, "Enter"); + expect(fetchMock).not.toHaveBeenCalled(); + }); + + it("triggers validation on Enter key press when API key is provided", async () => { + const fetchMock = vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ valid: true }), + }); + global.fetch = fetchMock as any; + + const el = render({}); + const input = el.querySelector('input[type="password"]'); + expect(input).toBeTruthy(); + + setInputValue(input!, "sk-test1234567890"); + dispatchKeyDown(input!, "Enter"); + + expect(fetchMock).toHaveBeenCalledWith( + "/api/providers/validate", + expect.objectContaining({ + method: "POST", + body: expect.stringContaining("sk-test1234567890"), + }) + ); + }); +}); diff --git a/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx b/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx new file mode 100644 index 000000000000..b86f4188c6a3 --- /dev/null +++ b/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx @@ -0,0 +1,92 @@ +// @vitest-environment jsdom +/** + * CheaperInferenceSponsorBanner — render gate (localStorage dismissal), CTA + * pointing at our link.omniroute.online branded short link, and discreet + * partner-link note. Mirrors kimiSponsorBanner.test.tsx, minus the version gate + * (this banner is a durable partnership, not a time-boxed offer). + */ +import React from "react"; +import { act } from "react"; +import { createRoot } from "react-dom/client"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; + +const STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; +const SHORT_URL = "https://link.omniroute.online/cheaper"; + +vi.mock("next-intl", () => ({ useTranslations: () => (k: string) => k })); +vi.mock("@/shared/components/ProviderIcon", () => ({ default: () => null })); + +async function renderBanner(): Promise { + vi.resetModules(); + const { default: CheaperInferenceSponsorBanner } = + await import("../../../src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner"); + + const container = document.createElement("div"); + document.body.appendChild(container); + const root = createRoot(container); + act(() => { + root.render(); + }); + return container; +} + +describe("CheaperInferenceSponsorBanner", () => { + beforeEach(() => { + ( + globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean } + ).IS_REACT_ACT_ENVIRONMENT = true; + localStorage.removeItem(STORAGE_KEY); + }); + + afterEach(() => { + document.body.innerHTML = ""; + localStorage.removeItem(STORAGE_KEY); + }); + + it("renders with the CTA pointing at the branded short link", async () => { + const container = await renderBanner(); + expect(container.textContent).toContain("title"); + expect(container.textContent).toContain("cta"); + const link = container.querySelector("a[href]"); + expect(link).not.toBeNull(); + expect(link?.getAttribute("href")).toBe(SHORT_URL); + expect(link?.getAttribute("target")).toBe("_blank"); + expect(link?.getAttribute("rel")).toContain("noopener"); + }); + + it("shows the discreet partner-link note near the CTA", async () => { + const container = await renderBanner(); + expect(container.textContent).toContain("partnerLinkNote"); + const link = container.querySelector("a[href]"); + expect(link?.getAttribute("title")).toBe("partnerLinkNote"); + }); + + it("hides after dismissal and stays hidden on re-render", async () => { + const first = await renderBanner(); + const button = first.querySelector("button"); + expect(button).not.toBeNull(); + act(() => { + button?.click(); + }); + expect(localStorage.getItem(STORAGE_KEY)).toBe("true"); + expect(first.textContent).not.toContain("title"); + + // a fresh render (simulating a later visit) stays hidden + const second = await renderBanner(); + expect(second.textContent).not.toContain("title"); + }); + + it("re-renders visible again only after the key is cleared", async () => { + const first = await renderBanner(); + const button = first.querySelector("button"); + act(() => { + button?.click(); + }); + expect(localStorage.getItem(STORAGE_KEY)).toBe("true"); + + localStorage.removeItem(STORAGE_KEY); + const second = await renderBanner(); + expect(second.textContent).toContain("title"); + }); +}); diff --git a/tests/unit/ui/providerPageHeaderKimiPartnerLink.test.tsx b/tests/unit/ui/providerPageHeaderKimiPartnerLink.test.tsx index dabc1266e044..519e6e03f210 100644 --- a/tests/unit/ui/providerPageHeaderKimiPartnerLink.test.tsx +++ b/tests/unit/ui/providerPageHeaderKimiPartnerLink.test.tsx @@ -49,7 +49,7 @@ describe("ProviderPageHeader — Kimi partner-link note", () => { it.each([ ["moonshot", "Kimi", "https://platform.kimi.ai?aff=omniroute"], ["kimi-coding", "Kimi Code CLI", "https://www.kimi.com/code?aff=omniroute"], - ["kimi-web", "Kimi Web", "https://www.kimi.com/code?aff=omniroute"], + ["kimi-web", "Kimi Web", "https://www.kimi.ai"], ])("flags the %s header link as a partner link", (id, name, website) => { const el = renderHeader(id, name, website); // The component also renders a "Back to Providers" above the diff --git a/tests/unit/ui/vscodeCopilotBanner.test.tsx b/tests/unit/ui/vscodeCopilotBanner.test.tsx index ed75ad4950a4..fad50b2b266a 100644 --- a/tests/unit/ui/vscodeCopilotBanner.test.tsx +++ b/tests/unit/ui/vscodeCopilotBanner.test.tsx @@ -11,7 +11,7 @@ import { createRoot } from "react-dom/client"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; const STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; vi.mock("next-intl", () => ({ useTranslations: () => (k: string) => k })); diff --git a/tests/unit/usage-command-json-format.test.ts b/tests/unit/usage-command-json-format.test.ts new file mode 100644 index 000000000000..9f76d6e19137 --- /dev/null +++ b/tests/unit/usage-command-json-format.test.ts @@ -0,0 +1,162 @@ +/** + * #8 (OmniCopilot) — the usage command answered `text/plain`, which a UI cannot + * parse safely. The structured form (`?format=json`) returns the same + * `ApiKeyUsageLimitStatus` + `UsageSnapshot` the text is rendered from. + * + * These tests pin the contract the extension depends on: JSON when asked, + * text by default, the 403 as a structured reason rather than a bare string, + * and an error body that never carries a stack trace. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { handleInternalUsageCommandHttpRequest } from "../../src/lib/usage/internalUsageCommand"; + +const NOW = Date.parse("2026-08-19T12:00:00.000Z"); + +const LIMIT_STATUS = { + enabled: true, + dailyLimitUsd: 5, + weeklyLimitUsd: 20, + dailySpentUsd: 1.25, + weeklySpentUsd: 8, + dailyWindowStartIso: "2026-08-19T03:00:00.000Z", + dailyResetAtIso: "2026-08-20T03:00:00.000Z", + weeklyWindowStartIso: "2026-08-16T03:00:00.000Z", + weeklyResetAtIso: "2026-08-23T03:00:00.000Z", + dailyExceeded: false, + weeklyExceeded: false, +}; + +function allowedDeps(overrides: Record = {}) { + return { + now: () => NOW, + isValidApiKey: async (apiKey: string) => apiKey === "sk-allowed", + getApiKeyMetadata: async () => ({ + id: "key-allowed", + name: "panel key", + allowUsageCommand: true, + usageLimitEnabled: true, + }), + getProviderConnections: async () => [ + { id: "conn-claude", provider: "claude", isActive: true }, + { id: "conn-codex", provider: "codex", isActive: true }, + ], + getAllProviderLimitsCache: () => ({ + "conn-claude": { + plan: "Claude Max", + quotas: { + weekly: { used: 25, total: 100, remaining: 75, resetAt: "2026-08-25T03:00:00.000Z" }, + }, + message: null, + fetchedAt: new Date(NOW).toISOString(), + }, + "conn-codex": { + plan: "Codex Pro", + quotas: { + weekly: { used: 9, total: 100, remaining: 91, resetAt: "2026-08-24T03:00:00.000Z" }, + }, + message: null, + fetchedAt: new Date(NOW).toISOString(), + }, + }), + getProviderConnectionById: async () => null, + getProviderLimitsCache: () => null, + getQuotaPolicy: async () => ({ defaultThresholdPercent: 0, providerWindowDefaults: {} }), + getApiKeyUsageLimitStatus: async () => LIMIT_STATUS, + ...overrides, + }; +} + +test("om-usage ?format=json returns the structured personal + provider quota", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + assert.match(response.headers.get("content-type") ?? "", /application\/json/); + const body = (await response.json()) as { + allowed: boolean; + personal: { dailySpentUsd: number } | null; + provider: { provider: string; connectionId: string } | null; + }; + assert.equal(body.allowed, true); + assert.equal(body.personal?.dailySpentUsd, 1.25); + assert.equal(body.provider?.provider, "claude"); + assert.equal(body.provider?.connectionId, "conn-claude"); +}); + +test("om-usage without ?format stays text/plain (the historical contract)", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + assert.match(response.headers.get("content-type") ?? "", /text\/plain/); + const text = await response.text(); + assert.match(text, /Personal quota/); + assert.match(text, /Provider quota/); +}); + +test("om-usage ?format=json returns every connection under providers[], not just the selected one", async () => { + // #11191 — a panel needs Codex + Claude side by side; the single `provider` + // pick is a terminal presentation choice, the collector had them all. + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + const body = (await response.json()) as { + allowed: boolean; + provider: { provider: string } | null; + providers: Array<{ provider: string }>; + }; + assert.equal(body.allowed, true); + const names = body.providers.map((s) => s.provider).sort(); + assert.deepEqual(names, ["claude", "codex"]); + // the single-pick field is still present and one of them + assert.ok(["claude", "codex"].includes(body.provider?.provider ?? "")); +}); + +test("om-usage ?format=json reports a disallowed key as structured allowed:false", async () => { + // A usage panel must tell "this key may not ask" apart from "no data yet", + // which a bare 403 text body cannot express. + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps({ + getApiKeyMetadata: async () => ({ id: "key-off", allowUsageCommand: false }), + }) + ); + + assert.equal(response.status, 403); + const body = (await response.json()) as { allowed: boolean }; + assert.equal(body.allowed, false); +}); + +test("om-usage ?format=json rejects an invalid key and never leaks a stack trace", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-wrong" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 401); + const body = (await response.json()) as { allowed: boolean; error?: { message?: string } }; + assert.equal(body.allowed, false); + assert.ok( + !body.error?.message?.includes("at /"), + "error bodies must not carry stack frames (ERROR_SANITIZATION)" + ); +}); diff --git a/tests/unit/validate-response-quality.test.ts b/tests/unit/validate-response-quality.test.ts index de4b5f47dc3a..6a918155e565 100644 --- a/tests/unit/validate-response-quality.test.ts +++ b/tests/unit/validate-response-quality.test.ts @@ -24,6 +24,20 @@ test("returns valid=true for SSE with 'data:' lines", async () => { assert.strictEqual(res.valid, true); }); +test("returns valid=true for SSE opening with a ':' comment line (e.g. OpenRouter keep-alive)", async () => { + const res = await validateResponseQuality( + makeResponse(': OPENROUTER PROCESSING\n\ndata: {"foo":"bar"}\n\n'), + false, + {} + ); + assert.strictEqual(res.valid, true); +}); + +test("returns valid=true for an SSE stream with leading whitespace before the first frame", async () => { + const res = await validateResponseQuality(makeResponse('\n\ndata: {"foo":"bar"}\n\n'), false, {}); + assert.strictEqual(res.valid, true); +}); + test("returns valid=false for non-JSON non-SSE text", async () => { const res = await validateResponseQuality(makeResponse("Hello world"), false, {}); assert.strictEqual(res.valid, false); diff --git a/tests/unit/webhook-discord-dispatcher.test.ts b/tests/unit/webhook-discord-dispatcher.test.ts index 4a49c3ba2ef0..605f4aa86b77 100644 --- a/tests/unit/webhook-discord-dispatcher.test.ts +++ b/tests/unit/webhook-discord-dispatcher.test.ts @@ -1,5 +1,6 @@ import test from "node:test"; import assert from "node:assert/strict"; +import { WEBHOOK_EVENT_VALUES } from "../../src/lib/webhooks/eventDescriptions.ts"; const { buildDiscordPayload } = await import("../../src/lib/webhooks/integrations/discord.ts"); @@ -14,15 +15,7 @@ test("buildDiscordPayload — request.failed produces embed with model", () => { }); test("buildDiscordPayload — all WEBHOOK_EVENTS return object with content or embeds", () => { - const events = [ - "request.completed", - "request.failed", - "provider.error", - "provider.recovered", - "quota.exceeded", - "combo.switched", - "test.ping", - ] as const; + const events = WEBHOOK_EVENT_VALUES; for (const event of events) { const payload = buildDiscordPayload(event, {}); assert.ok( @@ -33,7 +26,7 @@ test("buildDiscordPayload — all WEBHOOK_EVENTS return object with content or e }); test("buildDiscordPayload — embeds have title and color fields", () => { - const payload = buildDiscordPayload("provider.error", { provider: "openai" }); + const payload = buildDiscordPayload("request.failed", { provider: "openai" }); assert.ok(Array.isArray(payload.embeds) && payload.embeds.length > 0, "should have embeds"); const embed = payload.embeds![0]; assert.ok(typeof embed.title === "string" && embed.title.length > 0, "embed must have title"); diff --git a/tests/unit/webhook-slack-dispatcher.test.ts b/tests/unit/webhook-slack-dispatcher.test.ts index 44b34626cf28..ad91a774e38d 100644 --- a/tests/unit/webhook-slack-dispatcher.test.ts +++ b/tests/unit/webhook-slack-dispatcher.test.ts @@ -1,5 +1,6 @@ import test from "node:test"; import assert from "node:assert/strict"; +import { WEBHOOK_EVENT_VALUES } from "../../src/lib/webhooks/eventDescriptions.ts"; const { buildSlackPayload } = await import("../../src/lib/webhooks/integrations/slack.ts"); @@ -30,8 +31,8 @@ test("buildSlackPayload — test.ping produces a ping/test message", () => { ); }); -test("buildSlackPayload — provider.error includes provider context", () => { - const payload = buildSlackPayload("provider.error", { provider: "openai", model: "gpt-4" }); +test("buildSlackPayload — request.failed includes provider context", () => { + const payload = buildSlackPayload("request.failed", { provider: "openai" }); const combined = JSON.stringify(payload); assert.ok( combined.includes("Provider") || @@ -43,15 +44,7 @@ test("buildSlackPayload — provider.error includes provider context", () => { }); test("buildSlackPayload — all WEBHOOK_EVENTS produce valid payloads with text field", () => { - const events = [ - "request.completed", - "request.failed", - "provider.error", - "provider.recovered", - "quota.exceeded", - "combo.switched", - "test.ping", - ] as const; + const events = WEBHOOK_EVENT_VALUES; for (const event of events) { const payload = buildSlackPayload(event, {}); assert.ok( diff --git a/tests/unit/webhook-telegram-dispatcher.test.ts b/tests/unit/webhook-telegram-dispatcher.test.ts index eeb5ee2c92bf..29bf2f586c41 100644 --- a/tests/unit/webhook-telegram-dispatcher.test.ts +++ b/tests/unit/webhook-telegram-dispatcher.test.ts @@ -1,5 +1,6 @@ import test from "node:test"; import assert from "node:assert/strict"; +import { WEBHOOK_EVENT_VALUES } from "../../src/lib/webhooks/eventDescriptions.ts"; const { buildTelegramPayload, buildTelegramUrl } = await import("../../src/lib/webhooks/integrations/telegram.ts"); @@ -77,15 +78,7 @@ test("buildTelegramPayload — chat_id matches provided value for groups", () => }); test("buildTelegramPayload — all WEBHOOK_EVENTS produce valid payloads with chat_id", () => { - const events = [ - "request.completed", - "request.failed", - "provider.error", - "provider.recovered", - "quota.exceeded", - "combo.switched", - "test.ping", - ] as const; + const events = WEBHOOK_EVENT_VALUES; for (const event of events) { const payload = buildTelegramPayload(event, {}, "99999"); assert.equal(payload.chat_id, "99999"); diff --git a/tests/unit/webhooks-ghost-events.test.ts b/tests/unit/webhooks-ghost-events.test.ts index 427b0467a76e..a5d26e6019ce 100644 --- a/tests/unit/webhooks-ghost-events.test.ts +++ b/tests/unit/webhooks-ghost-events.test.ts @@ -30,4 +30,16 @@ describe("webhook catalogue", () => { const { notifyWebhookEvent } = await import("../../src/lib/webhookDispatcher.ts"); assert.equal(typeof notifyWebhookEvent, "function"); }); + + it("every builder accepts every value in WEBHOOK_EVENT_VALUES without throwing", async () => { + const { buildDiscordPayload } = await import("../../src/lib/webhooks/integrations/discord.ts"); + const { buildSlackPayload } = await import("../../src/lib/webhooks/integrations/slack.ts"); + const { buildTelegramPayload } = + await import("../../src/lib/webhooks/integrations/telegram.ts"); + for (const event of WEBHOOK_EVENT_VALUES) { + assert.doesNotThrow(() => buildDiscordPayload(event, {})); + assert.doesNotThrow(() => buildSlackPayload(event, {})); + assert.doesNotThrow(() => buildTelegramPayload(event, {}, "99999")); + } + }); }); diff --git a/tests/unit/zcode-executor.test.ts b/tests/unit/zcode-executor.test.ts index c82e7b336839..db4a8586bee9 100644 --- a/tests/unit/zcode-executor.test.ts +++ b/tests/unit/zcode-executor.test.ts @@ -26,6 +26,8 @@ function requestBody() { test("ZCode accepts GLM Coding Plan models and rejects unsafe/unknown ids", async () => { const { resolveZcodeModel } = await loadZcodeExecutor(); assert.deepEqual(resolveZcodeModel("glm-5.2"), { ok: true, model: "glm-5.2" }); + assert.equal(resolveZcodeModel("glm-5.2-high").ok, false); + assert.equal(resolveZcodeModel("glm-5.3-low").ok, false); assert.equal(resolveZcodeModel("-unexpected").ok, false); assert.equal(resolveZcodeModel("unknown-model").ok, false); }); @@ -70,7 +72,7 @@ test("ZCode buffers the completed turn into OpenAI SSE when stream=true", async }); const result = await executor.execute({ - model: "glm-5.2-high", + model: "glm-5.2", body: requestBody(), stream: true, credentials: {}, diff --git a/tests/unit/zcode-provider.test.ts b/tests/unit/zcode-provider.test.ts index 3acf3c862c4f..e882ab5fd91b 100644 --- a/tests/unit/zcode-provider.test.ts +++ b/tests/unit/zcode-provider.test.ts @@ -10,5 +10,18 @@ test("ZCode provider registry exposes a local no-auth GLM Coding Plan backend", assert.equal(zcodeProvider.baseUrl, "zcode://app-server/stdio"); assert.equal(zcodeProvider.authType, "none"); assert.equal(zcodeProvider.authHeader, "none"); - assert.equal(zcodeProvider.models.some((model) => model.id === "glm-5.2"), true); + assert.equal( + zcodeProvider.models.some((model) => model.id === "glm-5.2"), + true + ); + for (const alias of ["glm-5.3-high", "glm-5.3-low", "glm-5.2-high", "glm-5.2-max"]) { + assert.equal( + zcodeProvider.models.some((model) => model.id === alias), + false, + alias + ); + } + for (const model of zcodeProvider.models) { + assert.deepEqual(model.supportedThinkingEfforts, [], model.id); + } });