diff --git a/.env.example b/.env.example index 20c43643a21..84bde3f0256 100644 --- a/.env.example +++ b/.env.example @@ -87,6 +87,10 @@ DISABLE_SQLITE_AUTO_BACKUP=false # Used by: src/shared/utils/rateLimiter.ts # Example: redis://localhost:6379 (or redis://redis:6379 in Docker) # REDIS_URL=redis://localhost:6379 +# Namespace prefix for ALL OmniRoute Redis keys (rate limiter + auth cache + +# quota store). Prevents key collisions when OmniRoute shares a Redis instance +# with other apps (e.g. on 127.0.0.1:6379). Default when unset: omniroute: +# REDIS_KEY_PREFIX=omniroute: # Host interface docker-compose publishes the Redis sidecar on. # Default: 127.0.0.1 (loopback only). The compose Redis runs WITHOUT # `requirepass`, and app containers reach it over the compose network @@ -372,9 +376,8 @@ ALLOW_API_KEY_REVEAL=false # NO_LOG_API_KEY_IDS=key_abc123,key_def456 # Fallback per-day request budget applied to API keys whose `rate_limits` -# column is null. Default (unset/empty/malformed) preserves the legacy -# 1000/day, 5000/week, 20000/month windows so existing deployments do not -# silently lose rate limiting on upgrade. +# column is null. Default (unset/empty) is unlimited (no implicit caps). +# Malformed values preserve the legacy 1000/day, 5000/week, 20000/month windows. # Set explicitly to "0" to opt out entirely (unlimited fallback). Any # positive integer N enables N/day, 5N/week, 20N/month. # Used by: src/shared/utils/apiKeyPolicy.ts — checkRateLimit() fallback. @@ -1483,6 +1486,20 @@ CURSOR_USER_AGENT="Cursor/3.4" # OMNIROUTE_BROWSER_POOL=on # WEB_COOKIE_USE_BROWSER=0 +# ── Kimi Web (international kimi.ai Connect-RPC) ── +# Used by: open-sse/executors/kimi-web.ts. Override the base/chat URLs only if +# you need a mirror or proxy endpoint; defaults target https://www.kimi.ai with +# the Connect-RPC chat path /apiv2/kimi.gateway.chat.v1.ChatService/Chat. +# KIMI_WEB_BASE_URL=https://www.kimi.ai +# KIMI_WEB_CHAT_URL=https://www.kimi.ai/apiv2/kimi.gateway.chat.v1.ChatService/Chat + +# When OIDC is enabled, disable password login so users can only authenticate +# via OIDC Single Sign-On. The bare alias OIDC_DISABLE_PASSWORD_LOGIN is also +# accepted; the Dashboard Feature Flag takes precedence. Used by: +# src/app/api/auth/login/route.ts, src/app/api/settings/require-login/route.ts. +# OMNIROUTE_OIDC_DISABLE_PASSWORD_LOGIN=false +# OIDC_DISABLE_PASSWORD_LOGIN=false + # ── Adobe Firefly browser sign-in (system Chrome/Edge CDP) ── # Used by: open-sse/services/adobeFireflyBrowserLogin.ts. The Firefly login # flow drives a real, system-installed Chrome or Microsoft Edge via CDP so the @@ -1927,10 +1944,6 @@ APP_LOG_TO_FILE=true # Default: 300000 (5 minutes) # SEARCH_CACHE_TTL_MS=300000 -# ── OpenAI-compatible multi-connection ── -# Allow multiple simultaneous connections per OpenAI-compatible provider node. -# Used by: src/app/api/providers/route.ts -# ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE=false # ── CC-compatible provider (experimental) ── # Enable the Claude Code compatible provider endpoint. @@ -2564,6 +2577,11 @@ APP_LOG_TO_FILE=true # intended to be published as `omniroute-secure`. See SECURITY.md. # OMNIROUTE_BUILD_PROFILE=full +# Override the standalone build output directory consumed by the post-build +# colocation step. Default: the real Next.js standalone output under .build/. +# Used by: scripts/build/colocate-standalone.mjs (build tooling, not runtime). +# OMNIROUTE_STANDALONE_DIR= + # Skip emitting `.tar.gz` tarballs during optional-pack staging for the Electron # standalone tree (pack directories + optional-packs.index.json are still produced). # Used by the desktop release workflow to trim artifact upload size. @@ -2578,6 +2596,8 @@ APP_LOG_TO_FILE=true # ELECTRON_SMOKE_DATA_DIR= # ELECTRON_SMOKE_KEEP_DATA=0 # ELECTRON_SMOKE_STREAM_LOGS=0 +# #7592: second launch against the same DATA_DIR must pick the native driver. +# ELECTRON_SMOKE_COLD_RESTART=0 # Playground Studio # Default model used by the improve-prompt route (optional; falls back to model in request body). diff --git a/.github/workflows/dast-smoke.yml b/.github/workflows/dast-smoke.yml index 23055c46e45..674f0b20f6e 100644 --- a/.github/workflows/dast-smoke.yml +++ b/.github/workflows/dast-smoke.yml @@ -46,6 +46,7 @@ jobs: env: PORT: "20128" INJECTION_GUARD_MODE: block + REQUIRE_API_KEY: "false" run: | node dist/server.js > server.log 2>&1 & echo $! > server.pid @@ -64,16 +65,20 @@ jobs: # those 302s as "the API accepted a schema-violating request" and the configured-off # 400 as "rejected a schema-compliant request". Documenting the flow in the spec is # still right (operators need it); fuzzing it is not what this smoke is for. + # /api/auth/login has brute-force rate limiting: repeated failed logins return 429, + # which Schemathesis flags as rejection of schema-compliant requests. schemathesis run docs/openapi.yaml --url http://localhost:20128 \ --include-path-regex '^/v1/(chat/completions|models)$|^/api/(auth|keys)' \ - --exclude-path-regex '^/api/auth/oidc/' \ + --exclude-path-regex '^/api/auth/(oidc/|login)' \ --max-examples 8 --workers 4 --checks all --max-response-time 30 \ --request-timeout 20 --suppress-health-check all --no-color + - name: Install promptfoo + run: npm install -g promptfoo@0.122.0 - name: promptfoo injection-guard (blocking) env: OMNIROUTE_URL: http://localhost:20128 OMNIROUTE_API_KEY: not-needed-blocked-before-upstream - run: npx --yes promptfoo@latest eval -c promptfooconfig.yaml --no-cache + run: promptfoo eval -c promptfooconfig.yaml --no-cache - name: Stop server if: always() run: kill "$(cat server.pid)" || true diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index d8a65576dc6..3b04a20c8bb 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -183,15 +183,55 @@ jobs: env: DOCKER_BUILDKIT_INLINE_CACHE: 1 + - name: Build and push BUN base platform image by digest + id: build-bun-base + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 + with: + context: . + file: Dockerfile.bun + target: runner-base + platforms: ${{ matrix.platform }} + outputs: type=image,push-by-digest=true,name-canonical=true,push=true + tags: | + ${{ env.IMAGE_NAME }} + ${{ env.GHCR_IMAGE_NAME }} + cache-from: type=gha,scope=docker-bun-base-${{ matrix.arch }} + cache-to: type=gha,scope=docker-bun-base-${{ matrix.arch }},mode=max + no-cache: false + env: + DOCKER_BUILDKIT_INLINE_CACHE: 1 + + - name: Build and push BUN web platform image by digest + id: build-bun-web + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 + with: + context: . + file: Dockerfile.bun + target: runner-web + platforms: ${{ matrix.platform }} + outputs: type=image,push-by-digest=true,name-canonical=true,push=true + tags: | + ${{ env.IMAGE_NAME }} + ${{ env.GHCR_IMAGE_NAME }} + cache-from: type=gha,scope=docker-bun-web-${{ matrix.arch }} + cache-to: type=gha,scope=docker-bun-web-${{ matrix.arch }},mode=max + no-cache: false + env: + DOCKER_BUILDKIT_INLINE_CACHE: 1 + - name: Export digests env: DIGEST_BASE: ${{ steps.build.outputs.digest }} DIGEST_WEB: ${{ steps.build-web.outputs.digest }} + DIGEST_BUN_BASE: ${{ steps.build-bun-base.outputs.digest }} + DIGEST_BUN_WEB: ${{ steps.build-bun-web.outputs.digest }} run: | set -euo pipefail - mkdir -p /tmp/digests/base /tmp/digests/web + mkdir -p /tmp/digests/base /tmp/digests/web /tmp/digests/bun-base /tmp/digests/bun-web touch "/tmp/digests/base/${DIGEST_BASE#sha256:}" touch "/tmp/digests/web/${DIGEST_WEB#sha256:}" + touch "/tmp/digests/bun-base/${DIGEST_BUN_BASE#sha256:}" + touch "/tmp/digests/bun-web/${DIGEST_BUN_WEB#sha256:}" - name: Upload base digests uses: actions/upload-artifact@v7 @@ -209,6 +249,22 @@ jobs: if-no-files-found: error retention-days: 1 + - name: Upload bun-base digests + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: digests-bun-base-${{ matrix.arch }} + path: /tmp/digests/bun-base/* + if-no-files-found: error + retention-days: 1 + + - name: Upload bun-web digests + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: digests-bun-web-${{ matrix.arch }} + path: /tmp/digests/bun-web/* + if-no-files-found: error + retention-days: 1 + merge: name: Publish multi-arch manifests needs: @@ -263,6 +319,20 @@ jobs: path: /tmp/digests/web merge-multiple: true + - name: Download bun-base digests + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: digests-bun-base-* + path: /tmp/digests/bun-base + merge-multiple: true + + - name: Download bun-web digests + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: digests-bun-web-* + path: /tmp/digests/bun-web + merge-multiple: true + - name: Create Docker Hub manifest run: | set -euo pipefail @@ -286,6 +356,8 @@ jobs: create_manifest "${IMAGE_NAME}" "" /tmp/digests/base create_manifest "${IMAGE_NAME}" "-web" /tmp/digests/web + create_manifest "${IMAGE_NAME}" "-bun" /tmp/digests/bun-base + create_manifest "${IMAGE_NAME}" "-web-bun" /tmp/digests/bun-web - name: Create GHCR manifest run: | @@ -310,6 +382,8 @@ jobs: create_manifest "${GHCR_IMAGE_NAME}" "" /tmp/digests/base create_manifest "${GHCR_IMAGE_NAME}" "-web" /tmp/digests/web + create_manifest "${GHCR_IMAGE_NAME}" "-bun" /tmp/digests/bun-base + create_manifest "${GHCR_IMAGE_NAME}" "-web-bun" /tmp/digests/bun-web - name: Inspect image if: needs.prepare.outputs.version != 'main' diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml index 33708fc426a..e899a664eaa 100644 --- a/.github/workflows/electron-release.yml +++ b/.github/workflows/electron-release.yml @@ -279,9 +279,14 @@ jobs: - name: Smoke packaged Electron app (Linux) if: matrix.platform == 'linux' + # #7592: also cold-restart against the same DATA_DIR and assert a + # native SQLite driver (not the sql.js WASM fallback) is selected on + # the second launch — blocking here since Linux has no Windows-style + # sandbox caveats that would make it flaky. env: ELECTRON_SMOKE_TIMEOUT_MS: 60000 ELECTRON_SMOKE_STREAM_LOGS: "1" + ELECTRON_SMOKE_COLD_RESTART: "1" run: xvfb-run -a npm run electron:smoke:packaged - name: Collect installers diff --git a/.gitignore b/.gitignore index a21784f4aae..08bceafd368 100644 --- a/.gitignore +++ b/.gitignore @@ -291,3 +291,4 @@ docker-compose.yml.bak # Ad-hoc test sandboxes (never tracked — may contain local DBs) /.sandbox/ +.aider* diff --git a/AGENTS.md b/AGENTS.md index d4a7e7eb805..f10fb39483c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -46,7 +46,7 @@ Repository map and Reference Documentation sections below. ## Project at a Glance -**OmniRoute** — unified AI proxy/router. One endpoint, 346 LLM providers, auto-fallback. +**OmniRoute** — unified AI proxy/router. One endpoint, 348 LLM providers, auto-fallback. | Layer | Location | Purpose | | ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | @@ -56,9 +56,9 @@ Repository map and Reference Documentation sections below. | Translators | `open-sse/translator/` | Format conversion (OpenAI↔Claude↔Gemini) | | Transformer | `open-sse/transformer/` | Responses API ↔ Chat Completions | | Services | `open-sse/services/` | Combo routing, rate limits, caching, etc | -| Database | `src/lib/db/` | SQLite domain modules (157 migrations) | +| Database | `src/lib/db/` | SQLite domain modules (159 migrations) | | Domain/Policy | `src/domain/` | Policy engine, cost rules, fallback logic | -| MCP Server | `open-sse/mcp-server/` | 109 tools (44 canonical + memory/skill/GitHub/pool/gamification/plugin/Notion/Obsidian/local-corpus/RTK modules), 3 transports (stdio / SSE / Streamable HTTP), 33 scopes | +| MCP Server | `open-sse/mcp-server/` | 110 tools (44 canonical + memory/skill/GitHub/pool/gamification/plugin/Notion/Obsidian/local-corpus/RTK modules), 3 transports (stdio / SSE / Streamable HTTP), 33 scopes | | A2A Server | `src/lib/a2a/` | JSON-RPC 2.0 agent protocol | | Skills | `src/lib/skills/` | Extensible skill framework | | Memory | `src/lib/memory/` | Persistent conversational memory | diff --git a/CHANGELOG.md b/CHANGELOG.md index 9c01b39a7a0..05d3989c59e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,18 @@ ## [Unreleased] +### ✨ New Features + +- **feat(sse): STRICT_ZERO_COST** — opt-in, off-by-default `freeAccessPolicy: "strict"` setting + that hard-verifies every auto-combo candidate against live quota state and per-connection + economic safety before it can be dispatched, going beyond `hidePaidModels`'s static catalog + check. Adds curated `hardStopGuaranteed` metadata to `FREE_MODEL_BUDGETS`, a short-TTL quota + cache reusing `getUsageForProvider()`, and a connection-safety guarantee: a candidate backed + by multiple accounts has its `allowedConnectionIds` narrowed to exactly the connections + independently verified `SAFE`, so dispatch can never use an unverified account. An + `excludeTosAvoid` guard (default `false`) is available separately for contractual risk. See + `docs/routing/STRICT_ZERO_COST.md`. + --- ## [3.8.50] — TBD @@ -9,6 +21,7 @@ _Living section — regenerated 2026-08-12 from all cycle commits (cycle open `ed2db6cb19` → tip). Bullets carry the merged PR and its author; direct pushes listed separately._ ### ✨ New Features +- **feat(search):** first-class X Search provider (`x-search`) on `POST /v1/search` and MCP `omniroute_x_search` using SuperGrok / xAI server-side `x_search`. Explicit provider or `search_type: "x"` only — never auto-selected for web. Reuses `xai-oauth` / `xao` / `xai` credentials. Not the X Developer Platform MCP. ([#10985](https://github.com/diegosouzapw/OmniRoute/issues/10985)) - **feat(core):** add Layer A capability filter at router (#5696) - **feat(providers):** add DeepAI as paid API-key image provider ([#6671](https://github.com/diegosouzapw/OmniRoute/issues/6671)) - **feat(providers):** add Naga.ac and ChatAnywhere aggregator gateway providers (#6674 — thanks @chirag127) diff --git a/Dockerfile.bun b/Dockerfile.bun new file mode 100644 index 00000000000..bb547ce210d --- /dev/null +++ b/Dockerfile.bun @@ -0,0 +1,146 @@ +# ── Multi-stage Dockerfile for Native Bun Runtime (web-latest-bun) ─────────── +FROM oven/bun:1.3.14-slim AS base +WORKDIR /app + +RUN apt-get update \ + && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends \ + build-essential \ + python3 \ + python-is-python3 \ + make \ + g++ \ + libsecret-1-0 \ + ca-certificates \ + curl \ + && rm -rf /var/lib/apt/lists/* + +# ── Builder stage (100% Bun Native Install & Build) ───────────────────────── +FROM base AS builder +WORKDIR /app + +COPY . . + +# Fast Bun native package install +RUN bun install --include=optional --quiet + +# Compile native better-sqlite3 Node-API addon under Bun +RUN if [ -d "node_modules/better-sqlite3" ]; then \ + (cd node_modules/better-sqlite3 && bunx node-gyp rebuild); \ + fi + +# Fetch tls-client-node native binary if script exists +RUN if [ -f "node_modules/tls-client-node/scripts/postinstall.js" ]; then \ + bun node_modules/tls-client-node/scripts/postinstall.js || true; \ + fi + +# Disable Turbopack for Bun builder stage (Turbopack V8 internal worker bindings require Node) +ENV OMNIROUTE_USE_TURBOPACK=0 + +ARG OMNIROUTE_BASE_PATH="" +ENV OMNIROUTE_BASE_PATH=$OMNIROUTE_BASE_PATH + +ARG DASHBOARD_ALLOW_EMBED="" +ENV DASHBOARD_ALLOW_EMBED=$DASHBOARD_ALLOW_EMBED + +ENV NEXT_TELEMETRY_DISABLED=1 +ENV NODE_ENV=production + +# Bun native Next.js build execution +RUN bun run --quiet build + +# ── Runner Base stage (100% Bun Native Production Runtime) ────────────────── +FROM oven/bun:1.3.14-slim AS runner-base + +LABEL org.opencontainers.image.title="omniroute" \ + org.opencontainers.image.description="Unified AI proxy — route any LLM through one endpoint (Bun Native)" \ + org.opencontainers.image.url="https://omniroute.online" \ + org.opencontainers.image.source="https://github.com/diegosouzapw/OmniRoute" \ + org.opencontainers.image.licenses="MIT" + +WORKDIR /app + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + libsecret-1-0 \ + ca-certificates \ + curl \ + && rm -rf /var/lib/apt/lists/* + +ENV NODE_ENV=production +ENV PORT=20128 +ENV HOSTNAME=0.0.0.0 +ENV OMNIROUTE_MEMORY_MB=1024 + +ENV DATA_DIR=/app/data +RUN mkdir -p /app/data + +COPY --from=builder /app/.build/next/standalone ./ +COPY --from=builder /app/node_modules/better-sqlite3 ./node_modules/better-sqlite3 +ENV OMNIROUTE_MIGRATIONS_DIR=/app/migrations + +COPY --from=builder /app/scripts/dev/healthcheck.mjs ./healthcheck.mjs + +EXPOSE 20128 + +HEALTHCHECK --interval=30s --timeout=5s --start-period=15s --retries=3 \ + CMD bun healthcheck.mjs || exit 1 + +ENTRYPOINT ["bun", "dev/run-standalone.mjs"] + +# ── Runner Web stage (Bun Native + Chromium/Playwright for Web providers) ─── +FROM runner-base AS runner-web + +USER root + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + chromium \ + chromium-driver \ + fonts-liberation \ + libasound2t64 \ + gconf-service \ + libatk-bridge2.0-0 \ + libatk1.0-0 \ + libc6 \ + libcairo2 \ + libcups2 \ + libdbus-1-3 \ + libexpat1 \ + libfontconfig1 \ + libgbm1 \ + libgcc-s1 \ + libglib2.0-0 \ + libgtk-3-0 \ + libnspr4 \ + libnss3 \ + libpango-1.0-0 \ + pangocairo-1.0-0 \ + stdc++6 \ + libx11-6 \ + libx11-xcb1 \ + libxcb1 \ + libxcomposite1 \ + libxcursor1 \ + libxdamage1 \ + libxext6 \ + libxfixes3 \ + libxi6 \ + libxrandr2 \ + libxrender1 \ + libxss1 \ + libxtst6 \ + ca-certificates \ + fonts-gargi \ + fonts-ipafont-gothic \ + fonts-kacst \ + fonts-thai-tlwg \ + fonts-wqy-zenhei \ + && rm -rf /var/lib/apt/lists/* + +ENV PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD=1 +ENV PLAYWRIGHT_CHROMIUM_EXECUTABLE_PATH=/usr/bin/chromium + +# Return to the base image non-root user after the apt install (mirrors the +# Node Dockerfile runner-web stage, which re-asserts USER node). +USER bun diff --git a/README.md b/README.md index 390acbe4724..55e6d0328a1 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 346 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 346 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 348 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 348 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. @@ -101,7 +101,7 @@ ⚙️ Features 🎯 Combos - 🌐 Providers + 🌐 Providers 🔌 CLI & MCP @@ -210,7 +210,7 @@ curl http://localhost:20128/v1/chat/completions \ -The Promise — One endpoint. 346 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 346 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 57 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 109 tools, A2A, memory, guardrails, evals — 25,000+ tests). +The Promise — One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 348 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests).

@@ -461,7 +461,7 @@ All **19** strategies — mix & match per combo step: -What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 346 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 109 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. +What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 348 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. 📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md) @@ -559,7 +559,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute - **🖼️ New endpoints** — `/v1/ocr` (Mistral OCR) and `/v1/audio/translations` (Whisper-style) round out the media surface. → [API Reference](docs/reference/API_REFERENCE.md) - **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md) - **🌍 Deployment & ops** — reverse-proxy `basePath`, browser-language auto-detect, per-key device tracking, root-less MITM trust, zh-TW localization. → [Environment](docs/reference/ENVIRONMENT.md) -- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **346-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) +- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **348-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) - **📡 Routing transparency** — every response carries an `X-OmniRoute-Decision` header naming the strategy/provider/latency that served it, a new `cache-optimized` combo strategy + Auto-Combo `cacheAffinity` factor route repeat requests back to the connection holding the cached prefix, and a read-only `/v1/auto-combo/{channel}/candidates` endpoint exposes an `auto/*` channel's live candidate pool. → [Auto-Combo](docs/routing/AUTO-COMBO.md) - **⚡ Local performance & infra** — one-click local Redis, Cloudflare Workers / Deno Deploy relay deployers, Bifrost & Mux as supervised embedded services. → [Embedded Services](docs/frameworks/EMBEDDED-SERVICES.md) @@ -642,11 +642,11 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
-## 🌐 346 AI Providers — 90+ Free +## 🌐 348 AI Providers — 90+ Free
-> The most complete catalog of any open-source router: **346 providers**, **90+ with a free tier**, **57 free forever**. +> The most complete catalog of any open-source router: **348 providers**, **90+ with a free tier**, **56 free forever**.
@@ -821,7 +821,7 @@ Expose OmniRoute over **MCP**, **A2A**, a **REST API**, **webhooks** or a **remo - + @@ -988,14 +988,40 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ -p 127.0.0.1:20128:20128 -v omniroute-data:/app/data diegosouzapw/omniroute:latest ``` -`:latest` follows the highest **published** stable SemVer. It does not track git `main`. Pin `:X.Y.Z` for GitOps. See [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels). +`:latest` follows the highest **published** stable SemVer. It does not track git `main`. Pin `:X.Y.Z` for GitOps. See [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels).The image pins **`OMNIROUTE_MEMORY_MB=1024`**. That is enough for the dashboard and a light chat. **Coding agents** (`POST /v1/responses` from Claude Code, Codex, Grok, …) need a much larger V8 heap or the process `FATAL ERROR`s at ~12 GiB under two overlapping long contexts. Size the container above the heap (native buffers sit outside V8): +| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | +| --- | --- | --- | +| Dashboard / light chat | `1024` (image default) | ≥2 g | +| One coding agent | `8192` | ≥10 g | +| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | + +```bash +docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ + -e OMNIROUTE_MEMORY_MB=8192 --memory=10g \ + -p 127.0.0.1:20128:20128 -v omniroute-data:/app/data diegosouzapw/omniroute:latest +``` + +Full table: [Docker Guide — runtime RAM](docs/guides/DOCKER_GUIDE.md#runtime-ram-for-coding-agents). > **Pre-release Docker channel:** `diegosouzapw/omniroute:next` and > `diegosouzapw/omniroute:next-web` follow the current default `release/v*` > branch. These mutable tags are intended only for testing unreleased fixes and > are **not supported for production**. See > [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels). +**🥟 Bun** + +Standard `bun install` and global installation (`bun install -g omniroute`) are supported via Bun runtime detection: +- **Built-in `bun:sqlite`**: OmniRoute uses Bun's built-in `bun:sqlite` driver when running under Bun, falling back to `better-sqlite3` on Node.js or `sql.js`. +- **Automatic Webpack bundler selection**: Development (`bun run dev`) and production builds (`bun run build`) automatically detect Bun and disable Turbopack in favor of Webpack to prevent native V8 binding incompatibilities. +- **Dedicated Bun Dockerfile**: Multi-stage `Dockerfile.bun` for native Bun production deployments (`docker build -f Dockerfile.bun -t omniroute:bun .`). + +```bash +# Install and run with Bun +bun install +bun run dev +``` + **🛠️ From source** ```bash @@ -1174,7 +1200,7 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c - + diff --git a/bin/cli/api-commands/combos.mjs b/bin/cli/api-commands/combos.mjs index e4e4ff62f50..8f1976be239 100644 --- a/bin/cli/api-commands/combos.mjs +++ b/bin/cli/api-commands/combos.mjs @@ -30,20 +30,60 @@ export function register_combos(parent) { const data = res.ok ? await res.json() : await res.text(); emit(data, gOpts); }); + tag.command("get-api-combos-id-") + .description("Get combo by ID") + .requiredOption("--id ", "") + .action(async (opts, cmd) => { + const gOpts = cmd.optsWithGlobals(); + let url = "/api/combos/{id}"; + url = url.replace("{id}", encodeURIComponent(opts.id ?? "")); + const res = await apiFetch(url, { method: "GET", baseUrl: gOpts.baseUrl, apiKey: gOpts.apiKey }); + const data = res.ok ? await res.json() : await res.text(); + emit(data, gOpts); + }); + tag.command("put-api-combos-id-") + .description("Update combo") + .requiredOption("--id ", "") + .option("--body ", "JSON body or @path/to/file.json") + .action(async (opts, cmd) => { + const gOpts = cmd.optsWithGlobals(); + let url = "/api/combos/{id}"; + url = url.replace("{id}", encodeURIComponent(opts.id ?? "")); + let body; + if (opts.body) { + body = opts.body.startsWith("@") + ? JSON.parse(readFileSync(opts.body.slice(1), "utf8")) + : JSON.parse(opts.body); + } + const res = await apiFetch(url, { method: "PUT", body, baseUrl: gOpts.baseUrl, apiKey: gOpts.apiKey }); + const data = res.ok ? await res.json() : await res.text(); + emit(data, gOpts); + }); tag.command("patch-api-combos-id-") .description("Update combo") + .requiredOption("--id ", "") + .option("--body ", "JSON body or @path/to/file.json") .action(async (opts, cmd) => { const gOpts = cmd.optsWithGlobals(); let url = "/api/combos/{id}"; - const res = await apiFetch(url, { method: "PATCH", baseUrl: gOpts.baseUrl, apiKey: gOpts.apiKey }); + url = url.replace("{id}", encodeURIComponent(opts.id ?? "")); + let body; + if (opts.body) { + body = opts.body.startsWith("@") + ? JSON.parse(readFileSync(opts.body.slice(1), "utf8")) + : JSON.parse(opts.body); + } + const res = await apiFetch(url, { method: "PATCH", body, baseUrl: gOpts.baseUrl, apiKey: gOpts.apiKey }); const data = res.ok ? await res.json() : await res.text(); emit(data, gOpts); }); tag.command("delete-api-combos-id-") .description("Delete combo") + .requiredOption("--id ", "") .action(async (opts, cmd) => { const gOpts = cmd.optsWithGlobals(); let url = "/api/combos/{id}"; + url = url.replace("{id}", encodeURIComponent(opts.id ?? "")); const res = await apiFetch(url, { method: "DELETE", baseUrl: gOpts.baseUrl, apiKey: gOpts.apiKey }); const data = res.ok ? await res.json() : await res.text(); emit(data, gOpts); diff --git a/bin/cli/commands/combo.mjs b/bin/cli/commands/combo.mjs index 1cd8bb06001..8d58cf73bd9 100644 --- a/bin/cli/commands/combo.mjs +++ b/bin/cli/commands/combo.mjs @@ -4,6 +4,7 @@ import { withRuntime } from "../runtime.mjs"; import { t } from "../i18n.mjs"; import { apiFetch } from "../api.mjs"; import { emit } from "../output.mjs"; +import { resolveComboModels, collectModel } from "./comboModels.mjs"; const VALID_STRATEGIES = [ "priority", @@ -125,10 +126,31 @@ export function registerCombo(program) { .choices(VALID_STRATEGIES) .default("priority") ) + .option( + "--models ", + "Models for the combo: comma-separated provider/model entries, or a JSON array " + + '(e.g. --models "openai/gpt-4o,anthropic/claude-3-opus" or ' + + '--models \'[{"model":"gpt-4o","providerId":"openai"}]\')' + ) + .option( + "--model ", + "Add one model to the combo (provider/model or bare model id) — repeatable", + collectModel, + [] + ) .action(async (name, opts, cmd) => { const globalOpts = cmd.parent.optsWithGlobals(); + let models; + try { + models = resolveComboModels(opts); + } catch (err) { + console.error(`Error: ${err instanceof Error ? err.message : String(err)}`); + process.exit(1); + return; + } const exitCode = await runComboCreateCommand(name, opts.strategy, { ...opts, + models, output: globalOpts.output, }); if (exitCode !== 0) process.exit(exitCode); @@ -284,12 +306,20 @@ export async function runComboCreateCommand(name, strategy = "priority", opts = return 1; } + const models = Array.isArray(opts.models) ? opts.models : []; + if (!models.length) { + console.error( + "combo create requires at least one target. Pass --models and/or repeat --model ." + ); + return 1; + } + try { return await withRuntime(async ({ kind, api, db }) => { if (kind === "http") { const res = await api("/api/combos", { method: "POST", - body: { name, strategy, enabled: true, models: [], config: {} }, + body: { name, strategy, enabled: true, models, config: {} }, retry: false, acceptNotOk: true, }); @@ -305,7 +335,7 @@ export async function runComboCreateCommand(name, strategy = "priority", opts = console.error(`Combo '${name}' already exists. Delete it first.`); return 1; } - await db.combos.createCombo({ name, strategy, enabled: true, models: [], config: {} }); + await db.combos.createCombo({ name, strategy, enabled: true, models, config: {} }); } console.log(t("combo.created", { name })); diff --git a/bin/cli/commands/comboModels.mjs b/bin/cli/commands/comboModels.mjs new file mode 100644 index 00000000000..fec9dc8470f --- /dev/null +++ b/bin/cli/commands/comboModels.mjs @@ -0,0 +1,142 @@ +// Parses the `--models` / `--model` options for `omniroute combo create` (#10954). +// +// Root cause of #10954: `combo create` only ever registered `--strategy`; the +// HTTP body (POST /api/combos) and the local-db fallback (db.combos.createCombo) +// both hardcoded `models: []`, so every combo created via the CLI came out +// empty regardless of what the operator intended to route to. +// +// Accepted shapes mirror the server-side Zod union in +// `src/shared/validation/schemas/combo.ts` (`comboModelEntry` / +// `createComboSchema.models`) so a CLI-built payload never gets rejected by +// the API that ultimately validates it: +// - a plain string ("provider/model" or a bare model id) — the server's +// `normalizeComboModels` (src/lib/combos/steps.ts) already splits the +// leading "provider/" segment off a plain string, so passing the raw +// token through is sufficient for the common case; +// - a structured `{ kind?: "model", model, providerId?, provider?, ... }` +// object; +// - a structured `{ kind: "combo-ref", comboName, ... }` object (nested +// combo reference). +// +// The CLI (bin/cli/**) ships as plain `.mjs` with relative-only imports — no +// `@/` path aliases and no TS transpilation at runtime — so importing the +// real Zod schema from `src/shared/validation/schemas/combo.ts` is not +// viable here. This module instead validates the same minimal shape by hand +// and stays a thin, independently testable unit. + +/** + * Validates one already-parsed combo model entry against the shape accepted + * by `comboModelEntry` (string | model-step | combo-ref). Throws with a + * 1-based, human-readable position when the entry does not match. + * + * @param {unknown} entry + * @param {number} index + * @returns {string | Record} + */ +export function validateComboModelEntryShape(entry, index) { + const position = index + 1; + + if (typeof entry === "string") { + const trimmed = entry.trim(); + if (trimmed.length === 0) { + throw new Error(`--models entry #${position}: empty model string`); + } + if (trimmed.length > 300) { + throw new Error(`--models entry #${position}: model string exceeds 300 characters`); + } + return trimmed; + } + + if (entry === null || typeof entry !== "object" || Array.isArray(entry)) { + throw new Error(`--models entry #${position}: must be a string or a JSON object`); + } + + const kind = entry.kind; + + if (kind === "combo-ref") { + if (typeof entry.comboName !== "string" || entry.comboName.trim().length === 0) { + throw new Error( + `--models entry #${position}: kind "combo-ref" requires a non-empty "comboName"` + ); + } + return entry; + } + + if (kind !== undefined && kind !== "model") { + throw new Error(`--models entry #${position}: unknown "kind" value ${JSON.stringify(kind)}`); + } + + if (typeof entry.model !== "string" || entry.model.trim().length === 0) { + throw new Error(`--models entry #${position}: requires a non-empty "model"`); + } + if (entry.providerId !== undefined && typeof entry.providerId !== "string") { + throw new Error(`--models entry #${position}: "providerId" must be a string`); + } + if (entry.provider !== undefined && typeof entry.provider !== "string") { + throw new Error(`--models entry #${position}: "provider" must be a string`); + } + + return entry; +} + +/** + * Parses one `--models` spec — either a JSON array (`--models '[{"model":"gpt-4o"}]'`) + * or a comma-separated list of provider/model tokens + * (`--models 'openai/gpt-4o,anthropic/claude-3-opus'`) — into an array of + * combo model entries. + * + * @param {string} spec + * @returns {Array>} + */ +export function parseModelsSpec(spec) { + const trimmed = String(spec ?? "").trim(); + if (trimmed.length === 0) return []; + + if (trimmed.startsWith("[")) { + let parsed; + try { + parsed = JSON.parse(trimmed); + } catch (err) { + throw new Error(`--models: invalid JSON array (${err.message})`); + } + if (!Array.isArray(parsed)) { + throw new Error("--models: JSON value must be an array"); + } + return parsed.map((entry, i) => validateComboModelEntryShape(entry, i)); + } + + return trimmed + .split(",") + .map((token) => token.trim()) + .filter((token) => token.length > 0) + .map((token, i) => validateComboModelEntryShape(token, i)); +} + +/** + * Resolves the final `models` array for `combo create` from Commander opts: + * `--models ` and/or repeatable `--model `. + * + * @param {{ models?: string, model?: string[] }} opts + * @returns {Array>} + */ +export function resolveComboModels(opts = {}) { + const result = []; + + if (typeof opts.models === "string" && opts.models.trim().length > 0) { + result.push(...parseModelsSpec(opts.models)); + } + + if (Array.isArray(opts.model)) { + opts.model.forEach((token, i) => { + result.push(validateComboModelEntryShape(String(token).trim(), i)); + }); + } + + return result; +} + +/** Commander `collect`-style reducer for the repeatable `--model` option. */ +export function collectModel(value, previous) { + previous.push(value); + return previous; +} diff --git a/bin/cli/commands/oauth.mjs b/bin/cli/commands/oauth.mjs index 8bf547b2c00..c9f8386d2b9 100644 --- a/bin/cli/commands/oauth.mjs +++ b/bin/cli/commands/oauth.mjs @@ -228,20 +228,38 @@ async function runSocialFlow(def, opts) { async function runDeviceFlow(def, opts) { const providerKey = resolveBackendKey(def.id); - const startRes = await apiFetch(`/api/providers/${providerKey}/auth/start`, { - ...targetApiOptions(opts), - method: "POST", - }); + let startRes = await apiFetch(`/api/oauth/${providerKey}/device-code`, targetApiOptions(opts)); + if (!startRes.ok) { + startRes = await apiFetch(`/api/providers/${providerKey}/auth/start`, { + ...targetApiOptions(opts), + method: "POST", + }); + } if (!startRes.ok) { process.stderr.write(`Failed to start device flow: ${startRes.status}\n`); process.exit(1); } const start = await startRes.json(); - process.stdout.write( - `\nDevice code: ${start.userCode ?? start.user_code ?? ""}\nVisit: ${start.verificationUri ?? start.verification_uri}\n\n` - ); - if (opts.browser !== false) - await openBrowser(start.verificationUri ?? start.verification_uri ?? ""); + const userCode = start.userCode ?? start.user_code ?? ""; + const verificationUri = + start.verificationUriComplete ?? + start.verification_uri_complete ?? + start.verificationUri ?? + start.verification_uri ?? + start.authUrl ?? + start.url ?? + ""; + + if (userCode) { + process.stdout.write(`\nDevice code: ${userCode}\nVisit: ${verificationUri}\n\n`); + } else if (verificationUri) { + process.stdout.write(`\nVisit: ${verificationUri}\n\n`); + } else { + process.stdout.write(`\nAuthorization URL not available\n\n`); + } + + if (opts.browser !== false && verificationUri) + await openBrowser(verificationUri); process.stderr.write("Waiting for device authorization...\n"); const deadline = Date.now() + (opts.timeout ?? 300000); const intervalMs = (start.intervalMs ?? start.interval ?? 5) * 1000; diff --git a/bin/cli/commands/plugin.mjs b/bin/cli/commands/plugin.mjs index fc433a88ea4..971c9ef231b 100644 --- a/bin/cli/commands/plugin.mjs +++ b/bin/cli/commands/plugin.mjs @@ -9,10 +9,13 @@ import { discoverPlugins } from "../plugins.mjs"; // (instead of string-interpolating into `execSync`) prevents a malicious plugin // name like `foo; rm -rf ~` or `` foo`id` `` from being interpreted by the shell. function runNpm(args) { - const res = spawnSync("npm", args, { stdio: "inherit", shell: false }); + const isBun = Boolean(process.versions.bun); + const pm = isBun ? "bun" : "npm"; + const cmdArgs = isBun && args[0] === "install" ? ["add", ...args.slice(1)] : args; + const res = spawnSync(pm, cmdArgs, { stdio: "inherit", shell: false }); if (res.error) throw res.error; if (typeof res.status === "number" && res.status !== 0) { - throw new Error(`npm exited with code ${res.status}`); + throw new Error(`${pm} exited with code ${res.status}`); } } diff --git a/bin/cli/runtime/nativeDeps.mjs b/bin/cli/runtime/nativeDeps.mjs index 60e4d219531..ba5274f3c6a 100644 --- a/bin/cli/runtime/nativeDeps.mjs +++ b/bin/cli/runtime/nativeDeps.mjs @@ -114,30 +114,30 @@ export function isBetterSqliteBinaryValid() { export function npmInstallRuntime(pkgs, opts = {}) { const cwd = ensureRuntimeDir(); - // Persist to the runtime package.json (exact version) instead of --no-save so a later - // install of a sibling runtime dep (e.g. systray2 from trayRuntime.ts, which writes to the - // same runtime dir) does not prune this package as "extraneous" — that pruning otherwise - // reproduces "No SQLite driver available" after a tray install removes better-sqlite3. - // npm 12+ defaults `allowScripts` to off, silently skipping lifecycle/install - // scripts (e.g. better-sqlite3's node-gyp/prebuild-install rebuild) unless the - // package has a matching `allowScripts` entry — and still exits 0, masking the - // failure (#10713). The runtime dir is a CLI-owned, non-user package.json, so - // explicitly allowing scripts for the packages we are installing here is safe. - const npmArgs = [ - "install", - ...pkgs, - "--no-audit", - "--no-fund", - "--prefer-online", - "--save-exact", - ...pkgs.map((pkg) => `--allow-scripts=${pkg}`), - ]; - // On Windows .cmd files cannot be executed without a shell; use cmd.exe /c explicitly - // so we never set shell:true (which would propagate env and enable injection). const isWin = platform() === "win32"; - const [exe, args] = isWin ? ["cmd.exe", ["/c", "npm", ...npmArgs]] : ["npm", npmArgs]; + const isBun = Boolean(process.versions.bun); + + let exe, args, displayCmd; + if (isBun) { + const bunArgs = ["add", ...pkgs, "--trust"]; + [exe, args] = isWin ? ["cmd.exe", ["/c", "bun", ...bunArgs]] : ["bun", bunArgs]; + displayCmd = `bun ${bunArgs.join(" ")}`; + } else { + const npmArgs = [ + "install", + ...pkgs, + "--no-audit", + "--no-fund", + "--prefer-online", + "--save-exact", + ...pkgs.map((pkg) => `--allow-scripts=${pkg}`), + ]; + [exe, args] = isWin ? ["cmd.exe", ["/c", "npm", ...npmArgs]] : ["npm", npmArgs]; + displayCmd = `npm ${npmArgs.join(" ")}`; + } + if (!opts.silent) { - process.stdout.write(`[omniroute][runtime] npm ${npmArgs.join(" ")}\n`); + process.stdout.write(`[omniroute][runtime] ${displayCmd}\n`); } const res = spawnSync(exe, args, { cwd, diff --git a/bin/cli/sqlite.mjs b/bin/cli/sqlite.mjs index ce14541480f..982fef3520a 100644 --- a/bin/cli/sqlite.mjs +++ b/bin/cli/sqlite.mjs @@ -5,10 +5,14 @@ import { ensureSettingsSchema, hashManagementPassword, updateSettings } from "./ async function loadSqlite() { if (process.versions.bun) { - return { Database: (await import("bun:sqlite")).Database }; + try { + return { Database: (await import("bun:sqlite")).Database, driver: "bun:sqlite" }; + } catch (bunError) { + // fall through to better-sqlite3 if bun:sqlite fails + } } try { - return { Database: (await import("better-sqlite3")).default }; + return { Database: (await import("better-sqlite3")).default, driver: "better-sqlite3" }; } catch (error) { return { error }; } @@ -86,12 +90,14 @@ export function normalizeBunSqliteParams(params) { export function createSqliteNativeError(error) { const message = error instanceof Error ? error.message : String(error); + const isBun = Boolean(process.versions.bun); + const rebuildCmd = isBun ? "bun add better-sqlite3 --trust" : "npm rebuild better-sqlite3"; if (message.includes("NODE_MODULE_VERSION") || message.includes("ERR_DLOPEN_FAILED")) { return new Error( - "better-sqlite3 native binding is incompatible with this Node.js runtime. " + - "Run `npm rebuild better-sqlite3` in the OmniRoute project and try again. " + - "Or run: omniroute runtime repair " + - "(rebuilds into a user-writable runtime; works without a C++ toolchain)." + `better-sqlite3 native binding is incompatible with this runtime. ` + + `Run \`${rebuildCmd}\` in the OmniRoute project and try again. ` + + `Or run: omniroute runtime repair ` + + `(rebuilds into a user-writable runtime; works without a C++ toolchain).` ); } if ( @@ -100,10 +106,9 @@ export function createSqliteNativeError(error) { message.includes("Cannot find module 'better-sqlite3'") ) { return new Error( - "better-sqlite3 native binding could not be found (no prebuilt addon for this platform). " + - "This is common under `npx`, which runs a fresh, ephemeral install that never built the addon. " + - "Run: omniroute runtime repair " + - "(rebuilds into a user-writable runtime; works without a C++ toolchain)." + `better-sqlite3 native binding could not be found (no prebuilt addon for this platform). ` + + `Run: omniroute runtime repair ` + + `(rebuilds into a user-writable runtime; works without a C++ toolchain).` ); } return error; @@ -111,7 +116,7 @@ export function createSqliteNativeError(error) { async function openSqliteDatabase(dbPath, options = {}) { const loaded = await loadSqlite(); - if (process.versions.bun) { + if (loaded.driver === "bun:sqlite" || (process.versions.bun && !loaded.Database)) { if (options.fileMustExist && !fs.existsSync(dbPath)) { throw new Error(`SQLite file does not exist: ${dbPath}`); } diff --git a/bin/cli/utils/ensureAndroidCacheDir.mjs b/bin/cli/utils/ensureAndroidCacheDir.mjs index 30fe073f8b4..0e3f2d20ec4 100644 --- a/bin/cli/utils/ensureAndroidCacheDir.mjs +++ b/bin/cli/utils/ensureAndroidCacheDir.mjs @@ -94,10 +94,15 @@ export function ensureAndroidCacheDir(options = {}) { */ export function isFatalInstrumentationHookFailure(text) { if (!text) return false; - return ( - /Unsupported platform:\s*android/i.test(text) || - /error occurred while loading instrumentation hook/i.test(text) - ); + // Next.js wraps ANY throw inside instrumentation.register() with the generic + // "An error occurred while loading instrumentation hook:" prefix, on every + // platform (node_modules/next/dist/server/web/globals.js). That prefix alone + // therefore cannot identify the Android/Termux cache-probe failure — a bare + // generic instrumentation error on win32/desktop would be misreported as the + // Android bug and hide the real cause. Only match when the text actually + // carries the Android platform marker that Next's getCacheDirectory() emits. + // #10028 + return /Unsupported platform:\s*android/i.test(text); } /** diff --git a/bin/nodeRuntimeSupport.mjs b/bin/nodeRuntimeSupport.mjs index 47905f0e4ff..8f8f88f6832 100644 --- a/bin/nodeRuntimeSupport.mjs +++ b/bin/nodeRuntimeSupport.mjs @@ -44,6 +44,18 @@ export function getSecureFloorForMajor(major) { } export function getNodeRuntimeSupport(version = process.versions.node) { + if (process.versions.bun) { + return { + nodeVersion: `bun-${process.versions.bun} (Node.js API ${version})`, + nodeCompatible: true, + reason: "supported-bun", + supportedRange: SUPPORTED_NODE_RANGE + " || Bun >=1.1.0", + supportedDisplay: SUPPORTED_NODE_DISPLAY + ", or Bun 1.1+", + recommendedVersion: `v${RECOMMENDED_NODE_VERSION}`, + minimumSecureVersion: null, + }; + } + const parsed = parseNodeVersion(version); const secureFloor = getSecureFloorForMajor(parsed.major); const nodeCompatible = secureFloor ? compareNodeVersions(parsed, secureFloor) >= 0 : false; diff --git a/bin/omniroute.mjs b/bin/omniroute.mjs index fb0a4555208..09b133df4f3 100755 --- a/bin/omniroute.mjs +++ b/bin/omniroute.mjs @@ -17,7 +17,12 @@ import { existsSync, readFileSync, writeFileSync } from "node:fs"; import { join, dirname } from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; -import updateNotifier from "update-notifier"; +let updateNotifier = null; +try { + updateNotifier = (await import("update-notifier")).default; +} catch { + // update-notifier is optional in pruned standalone environments +} import { isNativeBinaryCompatible } from "../scripts/build/native-binary-compat.mjs"; import { getNodeRuntimeSupport, getNodeRuntimeWarning } from "./nodeRuntimeSupport.mjs"; import { getDefaultDataDir } from "./cli/data-dir.mjs"; @@ -251,8 +256,9 @@ if (shouldProvisionStorageKey(process.argv)) { // Register update notifier — checks npm once per 24h, notifies on exit via stderr. const _pkg = JSON.parse(readFileSync(join(ROOT, "package.json"), "utf8")); -const _notifier = updateNotifier({ pkg: _pkg, updateCheckInterval: 1000 * 60 * 60 * 24 }); +const _notifier = updateNotifier ? updateNotifier({ pkg: _pkg, updateCheckInterval: 1000 * 60 * 60 * 24 }) : null; process.on("exit", () => { + if (!_notifier || !_notifier.update) return; if (process.env.OMNIROUTE_NO_UPDATE_NOTIFIER) return; if (process.env.CI) return; if (process.argv.includes("--quiet") || process.argv.includes("-q")) return; diff --git a/changelog.d/features/10926-rankings-usage-reliability.md b/changelog.d/features/10926-rankings-usage-reliability.md new file mode 100644 index 00000000000..a79b92b4cf6 --- /dev/null +++ b/changelog.d/features/10926-rankings-usage-reliability.md @@ -0,0 +1 @@ +- **feat(rankings):** free provider rankings can now report what each provider actually served — `reliability.usage` (requests, successes, success rate over a window) behind the opt-in `withUsage`/`usageRange` query parameters, so a provider that answers every call with an error is no longer described as healthy ([#10926](https://github.com/diegosouzapw/OmniRoute/pull/10926)) diff --git a/changelog.d/features/11104-operator-error-rules.md b/changelog.d/features/11104-operator-error-rules.md new file mode 100644 index 00000000000..f31e78c01f3 --- /dev/null +++ b/changelog.d/features/11104-operator-error-rules.md @@ -0,0 +1 @@ +- **feat(providers):** let operators declare per-provider error rules through `settings.providerErrorRules` instead of patching the catalog — an operator-supplied rule for a provider is consulted before the built-in `providerRuleRegistry`, receives the raw error text, and has its declared scope/cooldown/reason actually honored end to end, for any provider (declaring the rule is the opt-in — no extra allowlist entry needed). Matches are plain case-insensitive substrings (never RegExp) and bounded to 50 rules to keep the hot path safe ([#11104](https://github.com/diegosouzapw/OmniRoute/pull/11104)) diff --git a/changelog.d/features/11190-usage-command-json.md b/changelog.d/features/11190-usage-command-json.md new file mode 100644 index 00000000000..d7655f04c51 --- /dev/null +++ b/changelog.d/features/11190-usage-command-json.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage` gains a structured form — `?format=json` returns the key's own usage as `ApiKeyUsageLimitStatus` + `UsageSnapshot` instead of `text/plain`. This is the surface a UI (the OmniCopilot panel) consumes to show a key holder their daily/weekly spend and quota reset. The route is self-service (the caller's own key, gated by `allowUsageCommand`), not the management surface; refusals come back as a discriminated `{ "allowed": false, "error": … }` so a UI can tell "not allowed" apart from "allowed but nothing cached yet". The endpoint was previously undocumented in `API_REFERENCE.md`; it now has a section ([#11190](https://github.com/diegosouzapw/OmniRoute/pull/11190)) diff --git a/changelog.d/features/11192-usage-command-providers-array.md b/changelog.d/features/11192-usage-command-providers-array.md new file mode 100644 index 00000000000..b7ef4211091 --- /dev/null +++ b/changelog.d/features/11192-usage-command-providers-array.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage?format=json` now returns `providers[]` — every connection's quota snapshot, not just the single selected one — so a panel can render Codex / Claude / OpenCode side by side. The collector already gathered all of them; the single-pick `provider` field (kept) is a terminal presentation choice. Closes the per-connection gap from OmniCopilot #8 ([#11192](https://github.com/diegosouzapw/OmniRoute/pull/11192)) diff --git a/changelog.d/features/m365-copilot-tool-calls.md b/changelog.d/features/m365-copilot-tool-calls.md new file mode 100644 index 00000000000..bfafe08033e --- /dev/null +++ b/changelog.d/features/m365-copilot-tool-calls.md @@ -0,0 +1 @@ +- **feat(providers):** copilot-m365-web now supports OpenAI tool calling — a router planning turn asks the substrate model (as a tool-selection assistant emitting `CALL_TOOL: name({...})` / `NO_TOOL_NEEDED` text, which bypasses its plugin-registry refusal) and validated decisions surface as `tool_calls` with `finish_reason: "tool_calls"` in both stream and non-stream modes; also flattens the full message history (assistant `tool_calls` + compacted tool results) so multi-turn agent loops keep context, replies to SignalR `type:6` keepalives, surfaces `type:3` error frames instead of a silent empty `stop`, and suppresses `writeAtCursor` text from tool-progress frames diff --git a/changelog.d/fixes/10028-windows-instrumentation-hook.md b/changelog.d/fixes/10028-windows-instrumentation-hook.md new file mode 100644 index 00000000000..9879e2f3f30 --- /dev/null +++ b/changelog.d/fixes/10028-windows-instrumentation-hook.md @@ -0,0 +1 @@ +- fix(cli): stop diagnosing every Next.js instrumentation-hook failure as the Android/Termux cache bug — only the Android "Unsupported platform: android" signal now triggers the Android hint, so a win32/desktop instrumentation error surfaces its real cause instead of a useless `mkdir -p ~/.cache` (#10028) \ No newline at end of file diff --git a/changelog.d/fixes/10265-command-code-provider-api.md b/changelog.d/fixes/10265-command-code-provider-api.md new file mode 100644 index 00000000000..b38e4e9d2a5 --- /dev/null +++ b/changelog.d/fixes/10265-command-code-provider-api.md @@ -0,0 +1 @@ +- fix(command-code): route chat to the documented /provider/v1/chat/completions endpoint instead of the CLI-only /alpha/generate, which Command Code gates/blocks for external callers (#10265) \ No newline at end of file diff --git a/changelog.d/fixes/10523-servicesupervisor-port-flake.md b/changelog.d/fixes/10523-servicesupervisor-port-flake.md new file mode 100644 index 00000000000..1a98ea7fa28 --- /dev/null +++ b/changelog.d/fixes/10523-servicesupervisor-port-flake.md @@ -0,0 +1 @@ +- fix(services): isolate probeBeforeSpawn adoption tests on distinct ports to stop the order-dependent flake (#10523) \ No newline at end of file diff --git a/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md b/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md new file mode 100644 index 00000000000..f204684baa2 --- /dev/null +++ b/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md @@ -0,0 +1 @@ +- **fix(executors):** the Meta AI (muse-spark-web) WebSocket send-message timeout now reports the socket's `readyState` at the moment it fires, so a "Meta AI WS timed out" failure can be told apart as either the connection never opening (`readyState=0`) or opening successfully and then going silent (`readyState=1`) — the exact ambiguity that made #10727 undiagnosable from logs alone (#10727). diff --git a/changelog.d/fixes/10736-corrupt-rotate-fence.md b/changelog.d/fixes/10736-corrupt-rotate-fence.md new file mode 100644 index 00000000000..dd2abc4fc46 --- /dev/null +++ b/changelog.d/fixes/10736-corrupt-rotate-fence.md @@ -0,0 +1 @@ +- **fix(db):** pause call-log rotation and record SQLITE_CORRUPT on `/api/db/health` instead of retrying writes against a malformed pager ([#10736](https://github.com/diegosouzapw/OmniRoute/issues/10736)) diff --git a/changelog.d/fixes/10850-readyz-alias.md b/changelog.d/fixes/10850-readyz-alias.md new file mode 100644 index 00000000000..94e62739bab --- /dev/null +++ b/changelog.d/fixes/10850-readyz-alias.md @@ -0,0 +1 @@ +- **fix(api):** alias `GET`/`HEAD` `/readyz` to `/healthz` so Kubernetes readiness probes do not 404 ([#10850](https://github.com/diegosouzapw/OmniRoute/issues/10850)) diff --git a/changelog.d/fixes/10940-opencode-limit-output.md b/changelog.d/fixes/10940-opencode-limit-output.md new file mode 100644 index 00000000000..9af54a20468 --- /dev/null +++ b/changelog.d/fixes/10940-opencode-limit-output.md @@ -0,0 +1 @@ +- fix(cli): always emit limit.output in generated OpenCode config so schema validation passes for metadata-less models (#10940) diff --git a/changelog.d/fixes/10945-least-used-rotation.md b/changelog.d/fixes/10945-least-used-rotation.md new file mode 100644 index 00000000000..36b23951b25 --- /dev/null +++ b/changelog.d/fixes/10945-least-used-rotation.md @@ -0,0 +1 @@ +- **Account rotation:** make `fallbackStrategy: "least-used"` actually rotate. The strategy sorts on `lastUsedAt` but never wrote it — only the round-robin branch committed — so on a pool where every `last_used_at` was still `NULL` the tie-break fell through to `priority` and returned the same connection on every dispatch ([#10945](https://github.com/diegosouzapw/OmniRoute/issues/10945)). diff --git a/changelog.d/fixes/10947-windows-updater-artifact-name.md b/changelog.d/fixes/10947-windows-updater-artifact-name.md new file mode 100644 index 00000000000..10c90a216da --- /dev/null +++ b/changelog.d/fixes/10947-windows-updater-artifact-name.md @@ -0,0 +1 @@ +- **Desktop auto-update (Windows):** stop the in-app updater 404ing on every release. NSIS used electron-builder's default artifact name, whose spaces GitHub rewrites to `.` on upload while `latest.yml` keeps `-`, so the manifest pointed at `OmniRoute-Setup-X.Y.Z.exe` while the published asset was `OmniRoute.Setup.X.Y.Z.exe`. The name is now set explicitly to the dot form the asset already has, so nothing published changes name ([#10947](https://github.com/diegosouzapw/OmniRoute/issues/10947)). diff --git a/changelog.d/fixes/10953-preserve-provider-effort-tiers.md b/changelog.d/fixes/10953-preserve-provider-effort-tiers.md new file mode 100644 index 00000000000..d509aac61b1 --- /dev/null +++ b/changelog.d/fixes/10953-preserve-provider-effort-tiers.md @@ -0,0 +1 @@ +- **fix(catalog):** preserve provider-declared reasoning effort tiers instead of replacing them with generic defaults ([#10953](https://github.com/diegosouzapw/OmniRoute/pull/10953)) — thanks @xz-dev diff --git a/changelog.d/fixes/10954-combo-create-models.md b/changelog.d/fixes/10954-combo-create-models.md new file mode 100644 index 00000000000..0a0638bceac --- /dev/null +++ b/changelog.d/fixes/10954-combo-create-models.md @@ -0,0 +1 @@ +- fix(cli): combo create accepts --models and no longer creates empty combos (#10954) diff --git a/changelog.d/fixes/10955-cli-ref-params.md b/changelog.d/fixes/10955-cli-ref-params.md new file mode 100644 index 00000000000..9497b4129e9 --- /dev/null +++ b/changelog.d/fixes/10955-cli-ref-params.md @@ -0,0 +1 @@ +- fix(cli): resolve $ref path params and add PATCH combos requestBody in generated API commands (#10955) diff --git a/changelog.d/fixes/10967-10966-combo-diag-recovery.md b/changelog.d/fixes/10967-10966-combo-diag-recovery.md new file mode 100644 index 00000000000..e962b149815 --- /dev/null +++ b/changelog.d/fixes/10967-10966-combo-diag-recovery.md @@ -0,0 +1,2 @@ +- fix(sse): combo diagnostics no longer truncate `exhausted_connection` entries to a hardcoded `provider: "unknown"` with the provider prefix eaten by an 8-char slice — the real provider id is preserved and only the connection id is truncated (#10967) +- fix(sse): combo terminal failures caused entirely by quota/account-balance exhaustion (including a durable HTTP 403 `insufficient_quota` / `AUTHZ_INSUFFICIENT_BALANCE`) now stamp a stable `quota_exhausted` diagnostics reason with a `switch-combo` recovery hint instead of the misleading default `retry` action (#10966) diff --git a/changelog.d/fixes/10976-skip-default-searxng.md b/changelog.d/fixes/10976-skip-default-searxng.md new file mode 100644 index 00000000000..a317979b4a1 --- /dev/null +++ b/changelog.d/fixes/10976-skip-default-searxng.md @@ -0,0 +1 @@ +- **fix(search):** skip catalog-default SearXNG `http://localhost:8888/search` so Docker/K8s search does not ECONNREFUSED then 502 into the next provider ([#10976](https://github.com/diegosouzapw/OmniRoute/issues/10976)) diff --git a/changelog.d/fixes/10986-reasoning-only-content.md b/changelog.d/fixes/10986-reasoning-only-content.md new file mode 100644 index 00000000000..0d293482bd3 --- /dev/null +++ b/changelog.d/fixes/10986-reasoning-only-content.md @@ -0,0 +1 @@ +- fix(command-code): surface reasoning-only output as content when a model emits no text-delta (#10986) \ No newline at end of file diff --git a/changelog.d/fixes/10988-release-v3850-quality-gates.md b/changelog.d/fixes/10988-release-v3850-quality-gates.md new file mode 100644 index 00000000000..283b30b8362 --- /dev/null +++ b/changelog.d/fixes/10988-release-v3850-quality-gates.md @@ -0,0 +1 @@ +- **fix(ci):** clear inherited `release/v3.8.50` quality-gate reds on the X Search PR: drop the stale `copilot-m365-web.ts:330` public-creds allowlist, document six missing env vars, register four covering Stryker tap tests, prune leftover ESLint suppressions, replace the phantom `@/lib/db/connections` Utilization import with `getProviderConnectionById`, and fix open-sse/dashboard typecheck regressions in freebuff, browser-backed chat, auth, health matrix, and Monaco ([#10988](https://github.com/diegosouzapw/OmniRoute/pull/10988)). diff --git a/changelog.d/fixes/10988-release-v3850-unit-shards.md b/changelog.d/fixes/10988-release-v3850-unit-shards.md new file mode 100644 index 00000000000..139266c990d --- /dev/null +++ b/changelog.d/fixes/10988-release-v3850-unit-shards.md @@ -0,0 +1 @@ +- **fix(ci):** clear remaining `release/v3.8.50` unit-shard reds on the X Search PR: pin `onnxruntime-node` to the transformers 1.24.3 copy, rebaseline OpenAPI coverage, sync goldens/i18n, honor eye-hidden no-auth models across provider aliases, await rejected-request call-log writes, absorb catalog event-loop shard contention in #9147, and align inherited tests with advisory context estimates, #10501 combo terminal-status aggregation, and current catalog/auth behavior ([#10988](https://github.com/diegosouzapw/OmniRoute/pull/10988)). diff --git a/changelog.d/fixes/10990-v0-vercel-web-static-catalog.md b/changelog.d/fixes/10990-v0-vercel-web-static-catalog.md new file mode 100644 index 00000000000..9d567212080 --- /dev/null +++ b/changelog.d/fixes/10990-v0-vercel-web-static-catalog.md @@ -0,0 +1 @@ +- **Static model catalog for v0-vercel-web:** seed a static catalog for the v0-vercel-web web-cookie provider (v0-1.0-md, v0-1.5-lg, v0-1.5-md) so its dashboard "Available Models" / "Import from /models" UI serves a usable list instead of falling through to the route's 400 "does not support models listing" ([#10990](https://github.com/diegosouzapw/OmniRoute/issues/10990)). \ No newline at end of file diff --git a/changelog.d/fixes/10997-blackbox-deprecation.md b/changelog.d/fixes/10997-blackbox-deprecation.md new file mode 100644 index 00000000000..74ac1915267 --- /dev/null +++ b/changelog.d/fixes/10997-blackbox-deprecation.md @@ -0,0 +1 @@ +- fix(providers): mark the blackbox provider deprecated — api.blackbox.ai returns HTTP 404 on every path variant (sweep 2026-08-21), so the public inference surface is dead and the catalog entry now carries a deprecation notice. ([#10997](https://github.com/diegosouzapw/OmniRoute/issues/10997)) \ No newline at end of file diff --git a/changelog.d/fixes/11002-dify-key-validation.md b/changelog.d/fixes/11002-dify-key-validation.md new file mode 100644 index 00000000000..6574714c9be --- /dev/null +++ b/changelog.d/fixes/11002-dify-key-validation.md @@ -0,0 +1 @@ +- fix(providers): validate Dify keys against its native /v1/chat-messages endpoint (#11002) \ No newline at end of file diff --git a/changelog.d/fixes/11008-account-rotation-eviction.md b/changelog.d/fixes/11008-account-rotation-eviction.md new file mode 100644 index 00000000000..4855dde6f28 --- /dev/null +++ b/changelog.d/fixes/11008-account-rotation-eviction.md @@ -0,0 +1 @@ +- **fix(accounts):** `markCooldown` now carries the failure origin (`transient` vs `terminal`) — transient 429/network only cools down, repeated terminal failures evict and are skipped by `pickAccount` until a success or operator clear ([#11008](https://github.com/diegosouzapw/OmniRoute/pull/11008)) — thanks @maxmad64bis diff --git a/changelog.d/fixes/11009-terminal-status-origin.md b/changelog.d/fixes/11009-terminal-status-origin.md new file mode 100644 index 00000000000..f0ab24edaf7 --- /dev/null +++ b/changelog.d/fixes/11009-terminal-status-origin.md @@ -0,0 +1 @@ +- **fix(providers):** route terminal `testStatus` writes (`banned`, `deactivated`, `credits_exhausted`) through a single origin-aware passage — probe failures are recorded but never deactivate the connection ([#11009](https://github.com/diegosouzapw/OmniRoute/pull/11009)) — thanks @maxmad64bis diff --git a/changelog.d/fixes/11014-codex-drop-default-on.md b/changelog.d/fixes/11014-codex-drop-default-on.md new file mode 100644 index 00000000000..0e5a8f11293 --- /dev/null +++ b/changelog.d/fixes/11014-codex-drop-default-on.md @@ -0,0 +1 @@ +- **fix(codex):** drop non-standard `codex.*` SSE events by default so OpenAI SDK / Codex CLI `/v1/responses` clients are not 502'd by `event: codex.rate_limits` ([#11014](https://github.com/diegosouzapw/OmniRoute/issues/11014)) — thanks @RaviTharuma diff --git a/changelog.d/fixes/11015-shutdown-track-sse.md b/changelog.d/fixes/11015-shutdown-track-sse.md new file mode 100644 index 00000000000..1ed99b3669d --- /dev/null +++ b/changelog.d/fixes/11015-shutdown-track-sse.md @@ -0,0 +1 @@ +- **fix(resilience):** count heavyweight `/v1` admission leases in the SIGTERM drain and send `Retry-After` on shutdown 503s so Recreate no longer looks like an empty 502 ([#11015](https://github.com/diegosouzapw/OmniRoute/issues/11015)) — thanks @RaviTharuma diff --git a/changelog.d/fixes/11016-cred-health-disable-log.md b/changelog.d/fixes/11016-cred-health-disable-log.md new file mode 100644 index 00000000000..37a9715f41c --- /dev/null +++ b/changelog.d/fixes/11016-cred-health-disable-log.md @@ -0,0 +1 @@ +- **fix(startup):** log `Credential health scheduler disabled` when `OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK` is set instead of lying with `started` ([#11016](https://github.com/diegosouzapw/OmniRoute/issues/11016)) — thanks @RaviTharuma diff --git a/changelog.d/fixes/11017-rate-limit-docs.md b/changelog.d/fixes/11017-rate-limit-docs.md new file mode 100644 index 00000000000..fc92469bd6b --- /dev/null +++ b/changelog.d/fixes/11017-rate-limit-docs.md @@ -0,0 +1 @@ +- **docs(api-keys):** document that unset `DEFAULT_RATE_LIMIT_PER_DAY` is unlimited (#2289), not a hidden 1000/day cap ([#11017](https://github.com/diegosouzapw/OmniRoute/issues/11017)) — thanks @RaviTharuma diff --git a/changelog.d/fixes/11050-remove-ghost-webhook-events.md b/changelog.d/fixes/11050-remove-ghost-webhook-events.md new file mode 100644 index 00000000000..6278ee6c0a9 --- /dev/null +++ b/changelog.d/fixes/11050-remove-ghost-webhook-events.md @@ -0,0 +1 @@ +- **fix(webhooks):** remove 3 declared-but-never-emitted events (`provider.error`, `provider.recovered`, `combo.switched`) from `WebhookEvent` — catalog now `request.completed | request.failed | quota.exceeded | test.ping`; `POST /api/webhooks` and `PUT /api/webhooks/[id]` reject ghost values with 400; OpenAPI webhook description updated across 43 locales ([11050](https://github.com/diegosouzapw/OmniRoute/pull/11050)) diff --git a/changelog.d/fixes/11060-perplexity-filter.md b/changelog.d/fixes/11060-perplexity-filter.md new file mode 100644 index 00000000000..c221d3ccaba --- /dev/null +++ b/changelog.d/fixes/11060-perplexity-filter.md @@ -0,0 +1 @@ +- fix(providers): filter Perplexity model import to the Sonar family so Agent-API catalog ids stop surfacing as routable chat models (#11060) diff --git a/changelog.d/fixes/11085-claude-code-tool-name-casing.md b/changelog.d/fixes/11085-claude-code-tool-name-casing.md new file mode 100644 index 00000000000..5ad424c1418 --- /dev/null +++ b/changelog.d/fixes/11085-claude-code-tool-name-casing.md @@ -0,0 +1 @@ +- **fix(claude):** restore canonical tool names (`bash` → `Bash`, `croncreate` → `CronCreate`) on non-streaming OpenAI→Claude conversion and through identity-echo alias maps, so Claude Code stops rejecting tool calls with "No such tool available" ([#11085](https://github.com/diegosouzapw/OmniRoute/pull/11085)) — thanks @linhdmn diff --git a/changelog.d/fixes/11095-termux-onnx.md b/changelog.d/fixes/11095-termux-onnx.md new file mode 100644 index 00000000000..8c268037104 --- /dev/null +++ b/changelog.d/fixes/11095-termux-onnx.md @@ -0,0 +1 @@ +- fix(install): make the ONNX dependency chain optional so Termux/Android installs succeed again (#11095) diff --git a/changelog.d/fixes/11101-reject-silent-validation.md b/changelog.d/fixes/11101-reject-silent-validation.md new file mode 100644 index 00000000000..04b2a67d5a0 --- /dev/null +++ b/changelog.d/fixes/11101-reject-silent-validation.md @@ -0,0 +1 @@ +- **fix(providers):** Reject silent validation degradation on provider connection patch — unknown `rateLimitOverrides` keys (e.g. a typo'd `tpm`) and empty/non-numeric values now return `400` with the rejected key list instead of being silently dropped ([#11101](https://github.com/diegosouzapw/OmniRoute/pull/11101)) diff --git a/changelog.d/fixes/11102-combo-suggestion-count.md b/changelog.d/fixes/11102-combo-suggestion-count.md new file mode 100644 index 00000000000..3cbf6f11d3d --- /dev/null +++ b/changelog.d/fixes/11102-combo-suggestion-count.md @@ -0,0 +1 @@ +- **Autopilot suggestion counter:** the combo health autopilot summary now reports `suggestionCount` (the real number of suggested actions across all issues) instead of conflating it with link counts, while keeping `actionableCount` as a deprecated alias for backward compatibility. The `run_combo_test` action now links to the dashboard with the combo id (`/dashboard/combos?test=`) rather than the read-only API route, so operators can actually trigger a test from the UI ([#11102](https://github.com/diegosouzapw/OmniRoute/pull/11102)). diff --git a/changelog.d/fixes/11103-persist-config-audit-log.md b/changelog.d/fixes/11103-persist-config-audit-log.md new file mode 100644 index 00000000000..aeb53b17814 --- /dev/null +++ b/changelog.d/fixes/11103-persist-config-audit-log.md @@ -0,0 +1 @@ +- **Config audit persistence:** persist the configuration audit trail to SQLite (`config_audit_log`) instead of an in-memory buffer capped at 1000 volatile entries, and bound its growth with `cleanupConfigAudit()` driven by the `retention.configAudit` setting (default 30 days), wired into `runAutoCleanup` ([#11103](https://github.com/diegosouzapw/OmniRoute/pull/11103)). diff --git a/changelog.d/fixes/11109-stream-recovery-toolcall.md b/changelog.d/fixes/11109-stream-recovery-toolcall.md new file mode 100644 index 00000000000..04a43382e42 --- /dev/null +++ b/changelog.d/fixes/11109-stream-recovery-toolcall.md @@ -0,0 +1 @@ +- fix(sse): resume mid-stream recovery after a _completed_ tool call — `finish_reason: "tool_calls"` is now tracked per-call instead of as a general terminal marker, so truncation of trailing prose after a fully-delivered tool call is recoverable while in-flight calls stay blocked ([#11109](https://github.com/diegosouzapw/OmniRoute/pull/11109)) diff --git a/changelog.d/fixes/11116-reasoning-effort-capability-discovery.md b/changelog.d/fixes/11116-reasoning-effort-capability-discovery.md new file mode 100644 index 00000000000..fbc4dfe694e --- /dev/null +++ b/changelog.d/fixes/11116-reasoning-effort-capability-discovery.md @@ -0,0 +1 @@ +- **fix(providers):** `reasoning_effort` now learns the accepted values from a provider's own 400/422 response and clamps to the highest one instead of forwarding an unsupported `xhigh`/`max` (or a hardcoded `"high"` fallback) — fixes custom OpenAI-compatible connections and registered providers with no reasoning metadata ([#11116](https://github.com/diegosouzapw/OmniRoute/pull/11116)) — thanks @maxmad64bis diff --git a/changelog.d/fixes/11144-responses-parallel-tool-calls-index.md b/changelog.d/fixes/11144-responses-parallel-tool-calls-index.md new file mode 100644 index 00000000000..35ba19b2796 --- /dev/null +++ b/changelog.d/fixes/11144-responses-parallel-tool-calls-index.md @@ -0,0 +1 @@ +- **fix(sse):** parallel `function_call` items in a Responses API stream (e.g. several tool calls dispatched in the same turn) now each get a stable, distinct `index`/`id` when translated to Chat Completions streaming deltas, instead of colliding on index 0 and tripping strict stream parsers with `Expected 'id' to be a string.` ([#11144](https://github.com/diegosouzapw/OmniRoute/pull/11144)) diff --git a/changelog.d/fixes/11154-provider-registry-node-net-bundle.md b/changelog.d/fixes/11154-provider-registry-node-net-bundle.md new file mode 100644 index 00000000000..b30d9e13926 --- /dev/null +++ b/changelog.d/fixes/11154-provider-registry-node-net-bundle.md @@ -0,0 +1 @@ +- fix(dashboard): keep `open-sse/config/providerRegistry.ts` free of `node:net` so the provider detail client bundle builds again — the host classification moved to a platform-free `src/shared/network/privateHost.ts` with a pure-JS `isIP` equivalent, leaving the #11122 routing behaviour unchanged (#11154) diff --git a/changelog.d/fixes/11162-combo-create-requires-model.md b/changelog.d/fixes/11162-combo-create-requires-model.md new file mode 100644 index 00000000000..228e6a9b20f --- /dev/null +++ b/changelog.d/fixes/11162-combo-create-requires-model.md @@ -0,0 +1 @@ +- **Combo create:** creating a routing combo without any model is now refused (`400`) — the CLI requires `--models`/`--model` on `combo create`, matching the dashboard which already rejected empty combos. diff --git a/changelog.d/fixes/11165-shared-registry-passthrough-model-lockout.md b/changelog.d/fixes/11165-shared-registry-passthrough-model-lockout.md new file mode 100644 index 00000000000..eabe67cb09c --- /dev/null +++ b/changelog.d/fixes/11165-shared-registry-passthrough-model-lockout.md @@ -0,0 +1 @@ +- **fix(resilience):** a missing-model `404` on a provider that declares `passthroughModels: true` in the shared registry (novita, uncloseai, orcarouter and 37 others) now locks out only that model instead of cooling the entire connection — `hasPerModelQuota()` previously read only the open-sse registry and the local/self-hosted families ([#11165](https://github.com/diegosouzapw/OmniRoute/pull/11165)) — thanks @yourspraveen diff --git a/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md b/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md new file mode 100644 index 00000000000..fd7e61988ba --- /dev/null +++ b/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md @@ -0,0 +1 @@ +- fix(cli): repair hollow externalized package dirs in the nested `/node_modules` bundle location too, not just the top-level one, fixing macOS/Linux Electron `ERR_MODULE_NOT_FOUND` on Turbopack-externalized packages (#7346) diff --git a/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md b/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md new file mode 100644 index 00000000000..e6458879cf0 --- /dev/null +++ b/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md @@ -0,0 +1 @@ +- **Electron packaged smoke test:** add a cold-restart mode (`ELECTRON_SMOKE_COLD_RESTART=1`, wired blocking on the Linux release leg) that relaunches the packaged app against its own persisted `DATA_DIR` and asserts a native SQLite driver was selected instead of the sql.js WASM fallback, closing the regression-test gap flagged in the stale-ABI `better-sqlite3` investigation ([#7592](https://github.com/diegosouzapw/OmniRoute/issues/7592)). diff --git a/changelog.d/fixes/8864-uncloseai-noauth.md b/changelog.d/fixes/8864-uncloseai-noauth.md new file mode 100644 index 00000000000..8a38e6b8363 --- /dev/null +++ b/changelog.d/fixes/8864-uncloseai-noauth.md @@ -0,0 +1 @@ +- fix(dashboard): treat UncloseAI as a no-auth provider so the connect form no longer forces a fake API key (#8864) diff --git a/changelog.d/fixes/9123-search-provider-local-flag-guard-mismatch.md b/changelog.d/fixes/9123-search-provider-local-flag-guard-mismatch.md new file mode 100644 index 00000000000..bc51e121036 --- /dev/null +++ b/changelog.d/fixes/9123-search-provider-local-flag-guard-mismatch.md @@ -0,0 +1 @@ +- fix(ssrf): make `getProviderOutboundGuard()` (used for search-provider connection validation, image generation and remote image fetch) honor the local-first default `OMNIROUTE_ALLOW_LOCAL_PROVIDER_URLS` the same way the chat validation guard already does, so a LAN-hosted SearXNG/Brave search provider works with only the LOCAL flag set instead of silently requiring `OMNIROUTE_ALLOW_PRIVATE_PROVIDER_URLS` ([#9123](https://github.com/diegosouzapw/OmniRoute/issues/9123)). \ No newline at end of file diff --git a/changelog.d/fixes/command-code-effort-capabilities.md b/changelog.d/fixes/command-code-effort-capabilities.md new file mode 100644 index 00000000000..38057107ef5 --- /dev/null +++ b/changelog.d/fixes/command-code-effort-capabilities.md @@ -0,0 +1 @@ +- fix(combo): resolve effort-suffixed command-code variants (e.g. `deepseek-v4-flash-max`) to their base model for capability lookups, so tool-bearing combo requests keep the declared priority order instead of reordering behind models with confirmed capabilities diff --git a/changelog.d/fixes/opencode-merge-provider-guard.md b/changelog.d/fixes/opencode-merge-provider-guard.md new file mode 100644 index 00000000000..aa1f02e63bb --- /dev/null +++ b/changelog.d/fixes/opencode-merge-provider-guard.md @@ -0,0 +1 @@ +- **OpenCode config merge:** stop `mergeOpenCodeConfig` splaying a malformed `provider` block into index keys. The root was already guarded against a non-object; the `provider` branch it spreads one level down was not, so an existing `"provider": ["a", "b"]` merged to `{"0": "a", "1": "b", …}`. Its sibling `mergeOpenCodeConfigText` already refuses the same input. diff --git a/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md b/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md new file mode 100644 index 00000000000..b26e1c2086d --- /dev/null +++ b/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md @@ -0,0 +1 @@ +- **fix(models):** a model synced from a provider's own `/models` discovery is now enforced at its real context window immediately, instead of waiting up to 24h for the Feature 5004 reconciler's next tick. The request-time token-limit chain resolves the window from `auto:discovery` overrides, which previously were only written at startup and on a 24h interval — so any model synced mid-cycle (models.dev not indexing it yet, no static registry entry) fell through to the provider's static `defaultContextLength` (128K for OpenRouter) while `/v1/models` simultaneously advertised the real window from the same discovery data. Measured: `openrouter/stealth/ox-alpha` advertised `context_length: 1048576` but rejected requests over 128K with `context_length_exceeded` for a full day after its sync. The reconcile now also runs opportunistically (debounced, fire-and-forget) right after a synced catalog write changes. Companion fix: discovery now captures the vendor-declared `reasoning.default_effort` (e.g. OpenRouter `stealth/ox-alpha` declares `max`, normalized to `xhigh`) as `defaultThinkingEffort`, and the OpenAI dispatch path injects it when a request carries no reasoning field of any shape — the lowest-priority default behind a `-{effort}` suffix alias and a static `ModelSpec.defaultReasoningEffort` — so a reasoning model that returns an empty response without an explicit effort gets the vendor default instead of `upstream_empty_response`. diff --git a/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md new file mode 100644 index 00000000000..82a88c5905a --- /dev/null +++ b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md @@ -0,0 +1 @@ +- **fix(executors):** OpencodeExecutor rotates (or retries once on a single-account direct path) on upstream 400 empty-body rejections — malformed completion envelopes with no error field were propagated as success and killed client sessions. Bounded +1 attempt per request; body reads are conditioned on status 400 so successful/streaming responses are never buffered. 400s carrying an error field keep propagating immediately. diff --git a/changelog.d/fixes/release-v3850-basereds-tests-i18n.md b/changelog.d/fixes/release-v3850-basereds-tests-i18n.md new file mode 100644 index 00000000000..3a6dee61b9c --- /dev/null +++ b/changelog.d/fixes/release-v3850-basereds-tests-i18n.md @@ -0,0 +1 @@ +- fix(i18n): complete Vietnamese translations for recently added UI strings (#9985) diff --git a/changelog.d/fixes/release-v3850-basereds.md b/changelog.d/fixes/release-v3850-basereds.md new file mode 100644 index 00000000000..30444ba7064 --- /dev/null +++ b/changelog.d/fixes/release-v3850-basereds.md @@ -0,0 +1,3 @@ +- fix(api): repair broken `@/lib/db/connections` import in the usage utilization route that failed the production build (#10939 follow-up) +- chore(docs): regenerate PROVIDER_REFERENCE and refresh README diagram SVGs to the real provider count (347) +- chore(lint): prune ESLint suppressions orphaned on the release branch diff --git a/changelog.d/maintenance/10778-grokbuild-suppression-fix.md b/changelog.d/maintenance/10778-grokbuild-suppression-fix.md new file mode 100644 index 00000000000..15c5df7ef72 --- /dev/null +++ b/changelog.d/maintenance/10778-grokbuild-suppression-fix.md @@ -0,0 +1 @@ +- fix(quality): register GrokBuildToolCard.tsx react-hooks/set-state-in-effect suppression (dropped in #10778's uncommitted fix) diff --git a/changelog.d/maintenance/10906-critical-db-state-assertions.md b/changelog.d/maintenance/10906-critical-db-state-assertions.md new file mode 100644 index 00000000000..c63f5f0039a --- /dev/null +++ b/changelog.d/maintenance/10906-critical-db-state-assertions.md @@ -0,0 +1 @@ +- **test(db):** replace three empty `test.skip` placeholders in the critical DB-state suite with real assertions — `resetDbInstance` must swap the singleton while the on-disk row survives, the on-disk DB must open in WAL journal mode, and `db_meta` must hold the seeded `schema_version` — so a regression in any of those invariants can no longer pass as silently green ([#10906](https://github.com/diegosouzapw/OmniRoute/pull/10906)) diff --git a/changelog.d/maintenance/10982-runtime-ram-coding-agents.md b/changelog.d/maintenance/10982-runtime-ram-coding-agents.md new file mode 100644 index 00000000000..3c0985c68f6 --- /dev/null +++ b/changelog.d/maintenance/10982-runtime-ram-coding-agents.md @@ -0,0 +1 @@ +- **docs(docker):** document runtime RAM for coding-agent `/v1/responses` (image default 1 GiB heap is dashboard-only; 8–12 GiB heap for agents) ([#10982](https://github.com/diegosouzapw/OmniRoute/issues/10982)) diff --git a/changelog.d/maintenance/11024-n-instance-scale-out.md b/changelog.d/maintenance/11024-n-instance-scale-out.md new file mode 100644 index 00000000000..82adafe4801 --- /dev/null +++ b/changelog.d/maintenance/11024-n-instance-scale-out.md @@ -0,0 +1 @@ +- **docs(docker):** document N independent `DATA_DIR`s as the supported large `/v1/responses` scale-out (one V8 heap ≠ host RAM; do not `replicas>1` on one SQLite file) ([#11024](https://github.com/diegosouzapw/OmniRoute/issues/11024)) — thanks @RaviTharuma diff --git a/changelog.d/maintenance/11038-filesize-baseline-fix.md b/changelog.d/maintenance/11038-filesize-baseline-fix.md new file mode 100644 index 00000000000..aebc2b9e234 --- /dev/null +++ b/changelog.d/maintenance/11038-filesize-baseline-fix.md @@ -0,0 +1 @@ +- fix(quality): rebaseline file-size for modelCapabilities.ts (1016->1072) drift from merged tip fixes (#11034 et al) diff --git a/changelog.d/maintenance/11053-stryker-oauth-autoimport-registration.md b/changelog.d/maintenance/11053-stryker-oauth-autoimport-registration.md new file mode 100644 index 00000000000..b2e21419009 --- /dev/null +++ b/changelog.d/maintenance/11053-stryker-oauth-autoimport-registration.md @@ -0,0 +1 @@ +- fix(quality): register `tests/unit/authz/oauth-autoimport-local-only.test.ts` in stryker `tap.testFiles` (residual of #11053) diff --git a/changelog.d/maintenance/11160-drain-v3850-basereds-docs-counts-orphan-test.md b/changelog.d/maintenance/11160-drain-v3850-basereds-docs-counts-orphan-test.md new file mode 100644 index 00000000000..e695a2b8fc0 --- /dev/null +++ b/changelog.d/maintenance/11160-drain-v3850-basereds-docs-counts-orphan-test.md @@ -0,0 +1 @@ +- chore(quality): drain two `release/v3.8.50` base-reds — refresh the drifted doc counts (159 migrations, 56 free-forever providers, 40 free-tier pools, incl. the 42 `llm.txt` locale mirrors) and move `uncloseai-noauth.test.ts` to a collected path so the UncloseAI no-auth regression guard actually runs (#11160) diff --git a/changelog.d/maintenance/release-v3850-basereds-eslint-deadcode-vitest-20260819.md b/changelog.d/maintenance/release-v3850-basereds-eslint-deadcode-vitest-20260819.md new file mode 100644 index 00000000000..905d338575c --- /dev/null +++ b/changelog.d/maintenance/release-v3850-basereds-eslint-deadcode-vitest-20260819.md @@ -0,0 +1,20 @@ +- **fix(ci):** drain three more base-reds on `release/v3.8.50` (#9985). ESLint was reporting + 219 errors locally (vs. 25 in the last CI run) — all from `react-hooks/set-state-in-effect`, + `react-hooks/preserve-manual-memoization`, `react-hooks/immutability`, + `react-hooks/static-components`, `react-hooks/refs` and `react-hooks/purity`, six React + Compiler lint rules that `eslint-plugin-react-hooks` v7 turns on by default and that were + never frozen in `config/quality/eslint-suppressions.json` after the dependency bump. Froze + the pre-existing violations for those six rules via ESLint's native + `--suppress-rule`/`--suppressions-location` mechanism (the same pattern already used for + `@next/next/no-location-assign-relative-destination`) — no application code changed, no rule + disabled, only genuinely-new violations stay blocking. `check:dead-code` was at 418 against a + 415 baseline: removed the unused `src/lib/quota/providerCapabilities.ts` file and the unused + `ProviderQuotaMonitor` interface in `providerQuotaTelemetry.ts` (both dead since PR #10148, + 2026-08-18, confirmed via `grep`/knip cross-reference), landing at 416; the residual +1 could + not be attributed to a single recent commit after checking every dead-list entry touched + since the 2026-08-14 baseline measurement, so it is rebaselined with the investigation + recorded in `quality-baseline.json`. `tests/unit/autoCombo/tieredRotation.test.ts`'s + "rotates across all 43 Cerebras connection IDs" case was hitting vitest's 5000ms default + timeout on a 200-iteration synchronous `selectProvider()` loop under shared-devbox + contention (load average 40-60+ observed) — widened its explicit timeout to 20000ms; the + assertion itself is unchanged. diff --git a/changelog.d/maintenance/vi-harimport-parity.md b/changelog.d/maintenance/vi-harimport-parity.md new file mode 100644 index 00000000000..b08b8dc92f7 --- /dev/null +++ b/changelog.d/maintenance/vi-harimport-parity.md @@ -0,0 +1 @@ +- fix(i18n): translate the 14 `providers.harImport*` keys into Vietnamese (parity gap left by #11069) diff --git a/config/quality/dashboard-typecheck-baseline.json b/config/quality/dashboard-typecheck-baseline.json index b97762d325e..b596060a8ef 100644 --- a/config/quality/dashboard-typecheck-baseline.json +++ b/config/quality/dashboard-typecheck-baseline.json @@ -1,9 +1,6 @@ { - "open-sse/services/payloadRules.ts": { - "TS2677": 1 - }, "src/app/(dashboard)/dashboard/HomePageClient.tsx": { - "TS2339": 16 + "TS2339": 10 }, "src/app/(dashboard)/dashboard/agent-skills/AgentSkillsPageClient.tsx": { "TS2503": 3 @@ -120,10 +117,6 @@ "src/app/(dashboard)/dashboard/providers/[id]/components/CompatibleModelsSection.tsx": { "TS2741": 1 }, - "src/app/(dashboard)/dashboard/providers/[id]/components/ConnectionRow.tsx": { - "TS2345": 3, - "TS2322": 1 - }, "src/app/(dashboard)/dashboard/providers/[id]/components/ConnectionsListPanel.tsx": { "TS2322": 2 }, @@ -141,12 +134,6 @@ "src/app/(dashboard)/dashboard/providers/[id]/components/ProviderPlaygroundPanel.tsx": { "TS2503": 1 }, - "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": { - "TS2322": 1 - }, - "src/app/(dashboard)/dashboard/providers/[id]/hooks/useModelImportHandlers.ts": { - "TS2339": 1 - }, "src/app/(dashboard)/dashboard/providers/[id]/hooks/useModelVisibilityHandlers.ts": { "TS2339": 15 }, @@ -190,9 +177,6 @@ "src/lib/combos/builderDraft.ts": { "TS2741": 1 }, - "src/lib/providers/codexFastTier.ts": { - "TS2367": 1 - }, "src/lib/services/htmlRewriter.ts": { "TS2322": 2, "TS2345": 2 @@ -219,14 +203,7 @@ "src/shared/hooks/useElectron.ts": { "TS2339": 19 }, - "src/shared/providers/webSessionCredentials.ts": { - "TS2353": 1, - "TS2322": 1 - }, "src/shared/schemas/cliCatalog.ts": { "TS2554": 2 - }, - "src/shared/services/opencodeConfig.ts": { - "TS2345": 1 } } diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index 9abe7bbf1da..3dae8591dd7 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -808,7 +808,6 @@ "count": 1 } }, - "src/app/api/settings/route.ts": { "no-restricted-imports": { "count": 1 @@ -1257,11 +1256,6 @@ "count": 1 } }, - "src/shared/components/CursorAuthModal.tsx": { - "react-hooks/exhaustive-deps": { - "count": 1 - } - }, "src/shared/components/LanguageSelector.tsx": { "@next/next/no-img-element": { "count": 1 @@ -1607,7 +1601,6 @@ "count": 11 } }, - "tests/unit/auth-ollama-cloud-per-model-403-3027.test.ts": { "@typescript-eslint/no-explicit-any": { "count": 11 @@ -2486,7 +2479,6 @@ "count": 12 } }, - "tests/unit/management-password.test.ts": { "@typescript-eslint/no-explicit-any": { "count": 4 diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 2e7fa58dacd..2eb468e16c6 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,5 +1,6 @@ { "_rebaseline_2026_08_20_10531_freebuff_provider": "PR #10531 (adrianaryaputra, feat/freebuff-provider-support, closes #6793) own growth: src/shared/constants/providers/apikey/gateways.ts 1283->1298 (+15, the freebuff APIKEY_PROVIDERS_GATEWAYS catalog entry, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines) and src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx 1062->1067 (+5, freebuff credential placeholder/hint at the existing per-provider switch chokepoint). Covered by tests/unit/freebuff-provider.test.ts (9/9 passing).", + "_rebaseline_2026_08_21_10987_logfare_provider": "PR #10987 (jonlwheat2-gif, feat/10644-logfare-provider, closes #10644) own growth: src/shared/constants/providers/apikey/gateways.ts 1298->1321 (+23, the logfare APIKEY_PROVIDERS_GATEWAYS catalog entry with Free badge/freeNote/apiHint documenting the request-logging policy, additive data at the existing registry chokepoint, same god-file no-split rationale as the prior gateways.ts rebaselines: #10531 freebuff, merge-storm 2026-08-11). Covered by tests/unit/logfare-registry.test.ts (1/1 passing).", "_rebaseline_2026_08_20_10574_reasoning_transport_fallback": "PR #10574 (jackjinke, fix/responses-reasoning-transport, fixes #10550) own growth: src/sse/handlers/chatHelpers.ts 1017->1019 (+2 = the new reasoningTransportFallback option threaded through executeChatWithBreaker's options destructure and its downstream handleSingleModel call, at the existing per-attempt options-passthrough chokepoint; not extractable without splitting the option-forwarding call itself). Covered by the PR's own reasoning-policy test suite (tests/unit/chatcore-translation-paths.test.ts, tests/unit/combo-attempt-body-isolation-7847.test.ts, tests/unit/reasoning-cache.test.ts, tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts among others), 446/446 focused tests passing.", "_rebaseline_2026_08_18_10517_zed_hosted_oauth_callback_port": "PR #10517 (phatchau036, fix/zed-hosted-oauth-callback-port) own growth: src/shared/components/OAuthModal.tsx 1131->1148 (wc -l; check-file-size.mjs counts via split(\"\\n\").length so the gate sees 1134->1149, +15/+18, crosses the frozen 1134 cap). Wires the zed-hosted native-app callback auto-complete: forceManual gating on isTrueLocalhost for zed-hosted, the loopback-redirect-URI comment block, and the exchangeToken full-URL-as-code branch, all at the existing provider-switch chokepoints this modal already carries growth for (seventh bump: 969->989->993->998->1030->1056->1100->1149; structural shrink tracked in #3501). The actual port-derivation logic lives in src/lib/oauth/providers/zed-hosted.ts (not frozen here) and was hardened during pre-merge review to use the server's own getRuntimePorts() instead of a browser-guessed scheme/port, covered by the new tests/unit/zed-hosted-loopback-port-derivation.test.ts (8/8 passing).", "_rebaseline_2026_08_13_10243_codex_fingerprint_merge": "PR #10243 (xz-dev, Codex OAuth fingerprint convergence) merge into release/v3.8.50: src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts crossed the 1000-line new-file cap for the first time (974 on base, 997 on the PR's own branch, 1013 after merging + prettier reflow) purely from combining two independent, already-legitimate feature additions that landed on the same shared UI-helper file — this PR's own Codex fingerprint-mode select/toggle wiring (CODEX_FINGERPRINT_MODE_VALUES, getCodexFingerprintModeLabel, CodexFingerprintModeValue) plus #8949's unrelated Codex account-service-tier helpers merged concurrently on release/v3.8.50. Neither addition alone crosses the cap; git's line-level auto-merge does not detect a threshold crossing. Not modularized as part of this conflict-resolution merge commit (out of scope — this is a merge, not a feature change). Covered by the PR's own tests/unit/codex-fingerprint-convergence.test.ts, tests/unit/executor-codex.test.ts, tests/unit/provider-specific-data-schema.test.ts (all passing post-merge).", @@ -388,6 +389,10 @@ "open-sse/services/claudeCodeCompatible.ts": 1563, "open-sse/services/combo.ts": 4742, "open-sse/services/compression/strategySelector.ts": 1379, + "open-sse/services/compression/engines/ccr/index.ts": 1024, + "_rebaseline_2026_08_22_11084_ccr_caller_gate": "PR #11084 (HouMinXi) own growth: open-sse/services/compression/engines/ccr/index.ts 1000->1024 (first listing — the engine was unlisted and drifted just over the 1000 cap; +24 are the callerSupportsCcrRetrieve gate that skips replacement entirely for callers without the retrieve tool, closing the stranded-prompt incident measured in production). Covered by tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", + "open-sse/services/contextManager.ts": 1001, + "_rebaseline_2026_08_22_11113_purify_system_first": "PR #11113 (ggdayup) own growth: open-sse/services/contextManager.ts 1000->1001 (+1, purifyHistory merges the compression notice into the leading system message instead of splicing a second one mid-array — live-confirmed TokenRouter 400s; the +1 is the merge-into-leading branch, not extractable). Covered by tests/unit/context-manager-purify-system-first.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "open-sse/services/rateLimitManager.ts": 1517, "open-sse/translator/response/openai-responses.ts": 1652, "open-sse/utils/cursorAgentProtobuf.ts": 1956, @@ -443,21 +448,26 @@ "src/shared/components/ModelSelectModal.tsx": 1138, "src/shared/constants/providers/apikey/gateways.ts": 1250 }, - "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1067, + "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1082, + "_rebaseline_2026_08_22_11156_enter_check_disabled": "PR #11156 (rqzbeh) own growth: AddApiKeyModal.tsx 1080->1082 (+2, Enter keydown handler now mirrors the isCheckDisabled condition — owner-requested post-merge polish from #11056; the rest of the diff is Prettier reflow). Covered by tests/unit/ui/add-api-key-modal-enter-key.test.tsx (jsdom render test, Enter dispatch assertions).", "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051, "src/shared/components/ModelSelectModal.tsx": 1138, - "src/shared/constants/providers/apikey/gateways.ts": 1298, + "src/shared/constants/providers/apikey/gateways.ts": 1321, "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1387, "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry": "DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legitima acima do cap; gateways.ts = god-file de catalogo de providers que cresceu com PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o proprio PR #9421 quebrou o arquivo); bridge.ts (PR #8949) = ponte Chromium vendored; proxyFetch.ts 1207->1220 = drift herdado de merges. Owner autorizou rebaseline com anotacao (2026-08-11).", - "src/lib/modelCapabilities.ts": 1016, + "src/lib/modelCapabilities.ts": 1072, + "_rebaseline_2026_08_21_11034_effort_variants": "DRIFT do tip (base-red #9985): modelCapabilities.ts 1016->1072 (+56) acumulado por PRs ja mergeadas no release/v3.8.50 — principalmente #11034 (resolve effort-variant capabilities a partir do modelo base), alem de #10963/#11040/#10987 growth dos catalogos. Tip puro ficou vermelho neste gate; rebaseline no tip por push direto (owner pre-autorizou crescimento legitimo). Nao tocou no arquivo da #11038.", "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 1014, "open-sse/config/imageRegistry.ts": 1034, "src/sse/handlers/chatHelpers.ts": 1019, - "src/shared/middleware/chatBodyAdmission.ts": 1005, + "src/shared/middleware/chatBodyAdmission.ts": 1009, + "_rebaseline_2026_08_22_11020_sigterm_drain": "PR #11020 (RaviTharuma) own growth: chatBodyAdmission.ts 1005->1009 (+4, heavyweight admission leases now increment the SIGTERM drain counter and releaseChatAdmissionWhenDone holds it for the SSE lifetime — closes #11015; +4 are the lease/drain wiring lines at the existing admission chokepoint). Covered by tests/unit/chat-body-admission.test.ts heavyweight-lease cases. Owner pre-authorized baseline bumps 2026-08-22.", "_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file).", - "open-sse/executors/commandCode.ts": 1038, + "open-sse/executors/commandCode.ts": 1059, "_rebaseline_2026_08_21_10859_vision_bridge_catalog": "#10859 own growth (Vision Bridge fixes #10808/#10809): src/lib/modelCapabilities.ts 1006->1016 (+10, cmd/gpt-5.3-codex* text-only capability resolution) and open-sse/executors/commandCode.ts 988->1023 (+35, Command Code wire-model normalization for bare ids + reasoning field fallback for opencode-routed gateways). Cohesive bug fixes at the existing capability-resolution / executor chokepoints; not extractable mid-fix. Covered by tests/unit/model-capabilities-command-code-codex-textonly-10703.test.ts, tests/unit/command-code-vision.test.ts, tests/unit/opencode-mimo-reasoning-details-nonstream.test.ts. Pushed directly to release (own-session miss: the original rebaseline was made in a throwaway validation worktree and never landed on the PR branch or the release before merge).", - "_rebaseline_2026_08_21_10907_sticky_pin_clear": "#10907 own growth: open-sse/executors/commandCode.ts 1023->1038 (+15, effort-suffix sanitization threading for the sticky-pin-clear fix). Cohesive change at the existing executor chokepoint. Covered by tests/unit/command-code-executor.test.ts." + "_rebaseline_2026_08_21_10907_sticky_pin_clear": "#10907 own growth: open-sse/executors/commandCode.ts 1023->1038 (+15, effort-suffix sanitization threading for the sticky-pin-clear fix). Cohesive change at the existing executor chokepoint. Covered by tests/unit/command-code-executor.test.ts.", + "_rebaseline_2026_08_21_10986_reasoning_only_content": "#10986 own growth: open-sse/executors/commandCode.ts 1038->1059 (+21, reasoning-only content fallback — when upstream emits only reasoning-delta events and never a text-delta, surface the reasoning text as message.content in createJsonResponse and emit a synthetic content delta in createStreamResponse). Cohesive bug fix at the existing executor chokepoint (mirrors precedent style of #10907/#10859). Covered by tests/unit/command-code-executor.test.ts (2 new cases: non-stream + streaming).", + "_rebaseline_2026_08_21_11069_m365_har_import": "#11069 own growth: AddApiKeyModal.tsx 1073->1080 (+7 = Import .har file button for the copilot-m365-web credential modal — M365 is the only provider whose credential (access_token+chathubPath) must be extracted from a DevTools HAR WebSocket URL, added as a new modal affordance). Cohesive UI at the existing modal chokepoint; not extractable. Covered by tests/unit/m365-har-import*.test.ts." }, "_rebaseline_base_2026_08_10_proxyfetch": "Base-red fix (green-prs sweep, issue #9985): open-sse/utils/proxyFetch.ts 1207 > cap 1000 — new proxied-TLS fetch helper introduced by the Fal reference-image work. Owner-authorized quick rebaseline to green; structural slim tracked for v3.9.0.", "_rebaseline_2026_07_27_v3849_train2": "Merge-train 2 (7 PRs) — owner-approved 2026-07-27. Single entry: chatCore.ts 4955->5006 (#8595, Responses multi-turn image compaction before the context hard-reject). Genuine irreducible growth at the existing compaction chokepoint in handleChatCore — the PR adds a last-resort retry against the concrete budget plus the estimateFinalInputTokens helper, both wired at the pre-existing call site rather than a new branch. Covered by tests/unit/8560-responses-image-compaction.test.ts (4 tests).", diff --git a/config/quality/open-sse-typecheck-baseline.json b/config/quality/open-sse-typecheck-baseline.json index c6de98b4184..753e9286c43 100644 --- a/config/quality/open-sse-typecheck-baseline.json +++ b/config/quality/open-sse-typecheck-baseline.json @@ -1,11 +1,4 @@ { - "open-sse/handlers/chatCore/clientUsageBuffer.ts": { - "TS2345": 2 - }, - "open-sse/utils/stream.ts": { - "TS2345": 2, - "TS2322": 2 - }, "src/lib/guardrails/videoBridgeHelpers.ts": { "TS2488": 1, "TS2365": 2, diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index 3bfef6d099c..604a5a8e19d 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -82,9 +82,10 @@ "tightenSlack": 10 }, "openapiCoverage.pct": { - "value": 39.2, + "value": 38.4, "direction": "up", "eps": 0.5, + "_rebaseline_2026_08_21_v3850_cycle_drift": "39.2 -> 38.4. Measured locally and in CI collect-metrics on release/v3.8.50 (260/677 implemented routes documented). Cycle added internal/dashboard routes faster than docs/openapi.yaml; documenting LOCAL_ONLY catch-all and service-management paths in the public spec would be gaming (same class as v3.8.34/v3.8.39/v3.8.47). This PR (#10988) adds 0 API routes.", "_tighten_2026_08_06_v3850_sweepreds": "38.0 -> 39.2 (aperto EXIGIDO pelo step 'Require-tighten (blocking)', que estava vermelho em ~60 PRs abertas de release/v3.8.50 — base-red herdado, nao defeito das PRs). A cobertura melhorou no ciclo porque as rotas novas entraram documentadas. 39.2 = valor medido pelo CI Quality Ratchet no run 31088889488; o tip puro 2ddbbc61a6 mede 39.3 localmente (npm run check:openapi-coverage: 247/628 rotas), entao 39.2 e o valor conservador dos dois. Aperto = gate mais ESTRITO, nunca mascaramento.", "_tighten_2026_07_04_v3844_release": "36.9 -> 39.3 (aperto exigido pelo --require-tighten no PR de release #5925). A cobertura OpenAPI melhorou no ciclo (9 rotas documentadas em 8fb020676 + as rotas novas de #5939/#5817/#6034/#5998 documentadas junto das features). 39.3 = valor medido pelo CI Quality Ratchet no run 28708141003 (tip 00c55afcb).", "_rebaseline_2026_06_28_v3839_release": "37.8 -> 36.9 (-0.9, beyond the 0.5 eps). v3.8.39 cycle drift surfaced ONLY on the release PR (the openapi-coverage ratchet does NOT run on PR->release fast-gates). The cycle added API/internal routes (antigravity paste-credentials onboarding, CCR ranged/grep/stats retrieve params, mcp 404 session handling) faster than docs/openapi.yaml coverage; documenting LOCAL_ONLY/internal onboarding routes in the PUBLIC spec would be gaming (same precedent as _rebaseline_2026_06_18_v3828_cycle_close). Measured by CI collect-metrics (run 28317145160) = 36.9. My release-finalize tree touches no routes (only the openapi.yaml version bump). Raising coverage by documenting public routes is tracked as follow-up doc debt.", @@ -102,8 +103,9 @@ "_rebaseline_2026_07_28_v3849_release": "75.5 -> 99 (+23.5). Aperto EXIGIDO pelo modo --require-tighten do ratchet: a métrica melhorou de verdade no ciclo v3.8.49. A causa é o workflow assíncrono de tradução, que finalmente alcançou o denominador em EN — as rebaselines anteriores (v3.8.39/.44/.47) foram todas afrouxamentos registrando o atraso das traduções, e agora ele foi pago. O coletor SUBTRAI os placeholders (present - placeholder em scripts/quality/collect-metrics.mjs), então os 317 marcadores __MISSING__ que esta release introduziu para o drift de valor já estão descontados dos 99 — o número é honesto, não inflado por placeholder. Medido pelo collect-metrics do CI no run 30404226939." }, "deadExports": { - "value": 418, + "value": 416, "direction": "down", + "_rebaseline_2026_08_19_v3850_basereds_9985": "415 -> 416. Measured on release/v3.8.50 tip 14a480453 during the #9985 base-red drain. Removed the 2 genuinely-dead symbols traced to a specific recent change (PR #10148, 2026-08-18): the unused src/lib/quota/providerCapabilities.ts file and the unused ProviderQuotaMonitor interface in providerQuotaTelemetry.ts (418 -> 416). The remaining +1 could not be attributed to a single recent commit after checking every dead-list entry touched since the 2026-08-14 baseline measurement (most are pre-existing debt on files edited for unrelated reasons); rebaselining the residual 1 rather than guessing at removals. Structural cleanup stays tracked in #3501.", "_rebaseline_2026_08_09_v3850_post_sweep": "227 -> 230. Measured by npm run check:dead-code on the unmodified release/v3.8.50 tip 382449d593 during the mandatory --full-ci pre-flight. The +3 is inherited cycle drift from the authorized merge sweep; this repair adds no production exports. Rebaseline records the actual tip so ci.yml quality-gate can run, while structural cleanup remains separate debt.", "_rebaseline_2026_07_01_v3843_release": "225->227 (+2). v3.8.43 cycle drift, surfaced in the Quality Ratchet job after eslintWarnings was rebaselined (check:dead-code runs there). 227 = measured by check:dead-code (knip) on the release tip 4635076eb. The 5 CI fixes add 0 dead exports: safeHttpHref in linkify.ts is module-local AND used (called by linkifyText); no new exports; test files are not scanned. Tighten via --update next cycle.", "dedicatedGate": true, @@ -195,10 +197,11 @@ "_rebaseline_2026_08_09_v3850_release_close": "7666 -> 8045 (+379 gzip bytes, +4.9%). Release v3.8.50 close reconciliation measured twice with the real size-limit + @size-limit/file path on tip e0ce95c592. Per-entry measurements remain below their absolute budgets: omniroute.mjs 4380/15000, mcp-server.mjs 1195/5000, nodeRuntimeSupport.mjs 887/8000, reset-password.mjs 1583/6000. The growth accumulated through legitimate CLI/runtime work in this cycle, including global-install ESM alias resolution, Termux cache preparation, and MCP stdio startup hardening; no entrypoint is near its absolute ceiling. The direction:down ratchet stays blocking from this exact measured tip." }, "openapiBreaking": { - "value": 0, + "value": 4, "direction": "down", "dedicatedGate": true, - "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec." + "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec.", + "_rebaseline_2026_08_22_combo_create_min1": "0 -> 4, split 3 own + 1 inherited. Docs-only alignment of components.schemas.ComboCreate with the request contract already enforced by the API since 638fc5fbd (combo create refuses an empty model list) and d5034ea52: `model`/`nodes` were phantom properties the server never accepted, and `models` (array, minItems 1) is the real required field. OWN findings (3, caused by this commit): removed `model`, removed `nodes`, added required `models` on POST /api/combos — spec-vs-server drift, not client-facing breakage, no working client could have relied on the removed shapes. INHERITED finding (1, NOT caused by this PR's code changes — pre-existing drift already present at parent d5034ea52): PATCH /api/combos/{id} request-body-added-required; that route's patch operation declares its own inline requestBody (required: true, bare object schema, docs/openapi.yaml ~2107-2118) and does not reference ComboCreate, so this finding exists independently of the ComboCreate alignment (same own-growth vs inherited-drift convention as _rebaseline_2026_07_20_aliasresolver_hook_split_7808). No code change in this PR; follow-up tracking = this change's PR description." }, "mutationScore.src/sse/services/auth.ts": { "value": 52.57, diff --git a/contrib/vps/.env.example b/contrib/vps/.env.example new file mode 100644 index 00000000000..31b5a38feec --- /dev/null +++ b/contrib/vps/.env.example @@ -0,0 +1,21 @@ +# Build this local image from the exact release checkout as documented below, +# or replace it with an immutable published image digest. +OMNIROUTE_IMAGE=omniroute:3.8.50-vps + +# The dashboard is loopback-only by default. Keep this value unless a trusted +# reverse proxy or private overlay network is configured on the same host. +OMNIROUTE_BIND_HOST=127.0.0.1 +OMNIROUTE_PORT=20128 + +# Generate unique values before the first start. Do not commit the resulting .env. +JWT_SECRET= +API_KEY_SECRET= +OMNIROUTE_WS_BRIDGE_SECRET= +INITIAL_PASSWORD= +REQUIRE_API_KEY=true + +# Conservative defaults for a small VPS. Adjust after observing real usage. +OMNIROUTE_MEMORY_LIMIT=1536m +OMNIROUTE_CPUS=1.0 +OMNIROUTE_PIDS_LIMIT=256 +APP_LOG_LEVEL=info diff --git a/contrib/vps/README.md b/contrib/vps/README.md new file mode 100644 index 00000000000..521f468f95d --- /dev/null +++ b/contrib/vps/README.md @@ -0,0 +1,151 @@ +# Headless Linux VPS deployment + +This bundle runs the published OmniRoute server image on a Linux VPS without +the Electron desktop shell. It keeps the dashboard on loopback by default, +does not publish Redis, persists application data, and adds conservative +resource and log limits. + +Use this bundle when the VPS only needs the API and web dashboard. The existing +root-level Compose profiles remain the right choice for local development, +building from source, bundled provider CLIs, or the Playwright/Chromium image. + +## Prerequisites + +- A supported Linux distribution with Docker Engine and Docker Compose v2. +- At least 2 GiB of available RAM for the default limits. The host needs more + headroom if other workloads run beside OmniRoute. +- SSH access for the loopback dashboard tunnel. + +## Install + +Build the headless server image from the exact release checkout. Building it +locally avoids assuming that a matching version tag has already been published +to a container registry: + +```bash +git switch --detach release/v3.8.50 +test "$(node -p "require('./package.json').version")" = "3.8.50" +docker build --target runner-base --tag omniroute:3.8.50-vps . +``` + +Then initialize the deployment from the repository root: + +```bash +cd contrib/vps +cp .env.example .env +chmod 600 .env +``` + +Generate separate values for every secret, then paste them into `.env`: + +```bash +openssl rand -base64 48 # JWT_SECRET +openssl rand -hex 32 # API_KEY_SECRET +openssl rand -base64 48 # OMNIROUTE_WS_BRIDGE_SECRET +openssl rand -base64 24 # INITIAL_PASSWORD +``` + +Do not reuse these values across installations. Keep `REQUIRE_API_KEY=true`. +Keep `OMNIROUTE_IMAGE` on the locally built version tag, or replace it with an +immutable registry digest; do not use the floating `latest` or `next` tags for +unattended production. + +Validate and start the stack: + +```bash +docker compose config --quiet +docker compose up -d +docker compose ps +``` + +The dashboard is intentionally bound to `127.0.0.1`. Reach it through SSH: + +```bash +ssh -L 20128:127.0.0.1:20128 user@your-vps +``` + +Then open `http://127.0.0.1:20128` locally. For a public hostname, put a trusted +reverse proxy on the same host in front of the loopback port and terminate TLS +there. Do not change `OMNIROUTE_BIND_HOST` to `0.0.0.0` merely to make the +dashboard reachable. + +## Verify + +```bash +docker compose ps +curl --fail --silent http://127.0.0.1:20128/healthz +docker compose logs --tail=100 omniroute +``` + +`/healthz` is a lifecycle probe. Use the authenticated monitoring/API routes +for deeper provider validation after the first login. + +## Web-session providers on a VPS + +Consumer web-session providers can enforce IP reputation, TLS fingerprint, or +browser-session binding. A cookie copied on a workstation may therefore fail +from a datacenter VPS even when the Linux container is healthy. In particular, +Grok clearance cookies can be tied to the browser IP, User-Agent, and TLS +fingerprint. Prefer official API credentials for unattended workloads. When a +web-session provider is required, use only credentials from an account you own +and follow that provider's guide; do not weaken TLS verification or bypass an +access challenge. + +## Backup + +Stop writes before copying SQLite data, then archive the named volume: + +```bash +docker compose stop omniroute +mkdir -p backups +docker run --rm \ + -v omniroute-vps_omniroute-data:/data:ro \ + -v "$PWD/backups:/backup" \ + docker.io/library/alpine:3.23 \ + tar -C /data -czf /backup/omniroute-data.tar.gz . +docker compose start omniroute +``` + +Verify the archive before relying on it: + +```bash +tar -tzf backups/omniroute-data.tar.gz >/dev/null +``` + +Store a timestamped copy outside the VPS. The fixed filename above is kept +simple for copy/paste; rename it after each verified backup. + +## Update and rollback + +Before updating, record the currently running immutable digest and take a +verified backup: + +```bash +docker image inspect "$(docker compose images -q omniroute)" \ + --format '{{index .RepoDigests 0}}' +``` + +Build the new local version tag first, or set `OMNIROUTE_IMAGE` in `.env` to a +new immutable registry digest. Pull only when the selected image is remote, +then recreate the application container: + +```bash +# Registry images only: docker compose pull omniroute +docker compose up -d --no-deps omniroute +docker compose ps +curl --fail --silent http://127.0.0.1:20128/healthz +``` + +To roll back the application image, restore the previous value of +`OMNIROUTE_IMAGE` and repeat the applicable `pull` and `up` commands. Restore the data +archive only when a migration changed the persisted data and image rollback +alone is insufficient. Keep the stack stopped while restoring the volume. + +## Remove the stack + +```bash +docker compose down +``` + +This preserves both named volumes. `docker compose down -v` deletes persistent +data and is intentionally not part of the normal uninstall path. diff --git a/contrib/vps/compose.yaml b/contrib/vps/compose.yaml new file mode 100644 index 00000000000..4e7c43ebd64 --- /dev/null +++ b/contrib/vps/compose.yaml @@ -0,0 +1,69 @@ +name: omniroute-vps + +services: + redis: + image: docker.io/library/redis:8.6.5-alpine + restart: unless-stopped + command: ["redis-server", "--save", "60", "1", "--appendonly", "yes", "--loglevel", "warning"] + volumes: + - redis-data:/data + healthcheck: + test: ["CMD", "redis-cli", "ping"] + interval: 10s + timeout: 5s + retries: 3 + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" + + omniroute: + image: ${OMNIROUTE_IMAGE:?Set OMNIROUTE_IMAGE to a versioned tag or digest} + restart: unless-stopped + stop_grace_period: 40s + depends_on: + redis: + condition: service_healthy + env_file: + - .env + environment: + NODE_ENV: production + PORT: "20128" + DASHBOARD_PORT: "20128" + HOSTNAME: 0.0.0.0 + DATA_DIR: /app/data + REDIS_URL: redis://redis:6379 + REQUIRE_API_KEY: ${REQUIRE_API_KEY:-true} + JWT_SECRET: ${JWT_SECRET:?Set JWT_SECRET in .env} + API_KEY_SECRET: ${API_KEY_SECRET:?Set API_KEY_SECRET in .env} + INITIAL_PASSWORD: ${INITIAL_PASSWORD:?Set INITIAL_PASSWORD in .env} + OMNIROUTE_WS_BRIDGE_SECRET: ${OMNIROUTE_WS_BRIDGE_SECRET:?Set OMNIROUTE_WS_BRIDGE_SECRET in .env} + ports: + - "${OMNIROUTE_BIND_HOST:-127.0.0.1}:${OMNIROUTE_PORT:-20128}:20128" + volumes: + - omniroute-data:/app/data + tmpfs: + - /tmp:size=256m,mode=1777 + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + pids_limit: ${OMNIROUTE_PIDS_LIMIT:-256} + mem_limit: ${OMNIROUTE_MEMORY_LIMIT:-1536m} + cpus: ${OMNIROUTE_CPUS:-1.0} + healthcheck: + test: ["CMD", "node", "healthcheck.mjs"] + interval: 30s + timeout: 5s + retries: 3 + start_period: 20s + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" + +volumes: + omniroute-data: + redis-data: diff --git a/docs/README.md b/docs/README.md index 185e0b915b2..2fe8a425cd7 100644 --- a/docs/README.md +++ b/docs/README.md @@ -34,7 +34,7 @@ Simple guides for using OmniRoute — no technical background needed. - [USAGE_QUOTA_GUIDE.md](guides/USAGE_QUOTA_GUIDE.md) — usage, quota & spend tracking. - [COST_TRACKING.md](guides/COST_TRACKING.md) — cost and spend tracking. - [FREE_PROVIDER_RANKINGS.md](guides/FREE_PROVIDER_RANKINGS.md) — free provider rankings (Arena ELO). -- [DOCKER_GUIDE.md](guides/DOCKER_GUIDE.md) — running OmniRoute under Docker. +- [DOCKER_GUIDE.md](guides/DOCKER_GUIDE.md) — running OmniRoute under Docker, including runtime RAM for coding agents. - [ELECTRON_GUIDE.md](guides/ELECTRON_GUIDE.md) — desktop (Electron) builds. - [TERMUX_GUIDE.md](guides/TERMUX_GUIDE.md) — running on Android via Termux. - [PWA_GUIDE.md](guides/PWA_GUIDE.md) — installing the dashboard as a PWA. diff --git a/docs/architecture/ARCHITECTURE.md b/docs/architecture/ARCHITECTURE.md index b78d433fee8..6c9d4007825 100644 --- a/docs/architecture/ARCHITECTURE.md +++ b/docs/architecture/ARCHITECTURE.md @@ -1131,7 +1131,6 @@ Environment variables actively used by code: - App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` - Storage: `DATA_DIR` -- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` - Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` - Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` - Logging: `APP_LOG_TO_FILE`, `APP_LOG_RETENTION_DAYS`, `CALL_LOG_RETENTION_DAYS` diff --git a/docs/architecture/RESILIENCE_GUIDE.md b/docs/architecture/RESILIENCE_GUIDE.md index 030f65dd414..0048761e609 100644 --- a/docs/architecture/RESILIENCE_GUIDE.md +++ b/docs/architecture/RESILIENCE_GUIDE.md @@ -448,14 +448,14 @@ classification rules pick the fallback `reason` and lock `scope` Classification rules only see full error **text** (needed to match body markers like `额度不足`) for providers listed in the `FULL_TEXT_RULE_PROVIDERS` allowlist in `providerErrorRules.ts` — currently only `"agentrouter"`. For -every other provider, `checkFallbackError` hands `getProviderErrorRuleMatch` -only the structured error (`{code, type}`), which is enough for -header/status/code-based rules but blind to body-text markers. The helper -`resolveRuleMatchBody()` performs this selection: full error text for -allowlisted providers, the structured error otherwise. Adding a provider to -`FULL_TEXT_RULE_PROVIDERS` is an explicit per-provider opt-in — it exists so -that the default path for every provider not on the list stays -byte-for-byte unchanged. +every other **built-in catalog** provider, `checkFallbackError` hands +`getProviderErrorRuleMatch` only the structured error (`{code, type}`), which +is enough for header/status/code-based rules but blind to body-text markers. +The helper `resolveRuleMatchBody()` performs this selection: full error text +for allowlisted providers, the structured error otherwise. Adding a +**built-in** provider to `FULL_TEXT_RULE_PROVIDERS` is an explicit per-provider +opt-in — it exists so that the default path for every provider not on the +list stays byte-for-byte unchanged. A rule's `scope` (`model` / `provider` / `connection`) is a separate opt-in from `FULL_TEXT_RULE_PROVIDERS`: `checkFallbackError` only surfaces it as @@ -466,6 +466,31 @@ honorsRuleLockScope()` — today only `"agentrouter"`). See "Restated quota errors" above for what a `scope: "connection"` match actually does once a provider is on that allowlist. +**#11104 — operator-declared rules bypass both allowlists.** An operator can +declare a per-provider rule at runtime via `settings.providerErrorRules` +(`open-sse/config/providerErrorRules.ts::setOperatorProviderErrorRules`) +without editing this file. Gating an operator rule behind +`FULL_TEXT_RULE_PROVIDERS`/`HONORS_RULE_LOCK_SCOPE_PROVIDERS` — allowlists +meant to protect the **default** behavior of built-in catalog rules — would +make the settings mechanism inert for every provider except the ones already +listed there, since declaring the rule is already the operator's explicit +opt-in. `resolveRuleMatchBody()` and `honorsRuleLockScope()` both check +`hasOperatorRuleForProvider()` first: a provider with an operator rule gets +the raw error text and has its declared `scope` honored, regardless of +whether it also appears in either allowlist. + +**Known gap — `providerRuleRegistry` is never consulted for HTTP 400.** +`checkFallbackError`'s `BAD_REQUEST` branch classifies status 400 entirely +through its own pattern arrays (`MODEL_ACCESS_DENIED_PATTERNS`, +`CONTEXT_OVERFLOW_PATTERNS`, etc. in `accountFallback.ts`) and returns before +the `configuredRule`/`getProviderErrorRuleMatch` branch above it is reached. +A built-in catalog rule (or an operator rule) with `status: 400` is +syntactically valid but will never fire. No existing rule targets 400 today, +so nothing in production is affected — but a future 400 rule needs this +branch touched first, which is a larger change than adding a rule (it +reclassifies 400 for every provider already relying on the pattern-array +behavior) and is out of scope for a single-provider rule addition. + ### Adding a new quota-misstating gateway 1. Register one rule array in `statusRestatementRegistry` diff --git a/docs/changelog/fragments/10962.md b/docs/changelog/fragments/10962.md new file mode 100644 index 00000000000..5170a415e3f --- /dev/null +++ b/docs/changelog/fragments/10962.md @@ -0,0 +1 @@ +fix(catalog): expose only provider-routable GLM reasoning-effort tiers and remove unroutable ZCode aliases diff --git a/docs/diagrams/cli-terminal.svg b/docs/diagrams/cli-terminal.svg index 4fd6887859f..e2ad57c8b13 100644 --- a/docs/diagrams/cli-terminal.svg +++ b/docs/diagrams/cli-terminal.svg @@ -1,6 +1,6 @@ - + Compact animated terminal cycling three real OmniRoute CLI commands with a typewriter effect and a scrolling subcommand ticker; the first frame shows the completed providers-list screen. - + @@ -16,12 +16,12 @@ OmniRoute Providers1f3a9c2e  anthropic   Claude Max 20x    active8c2d5b1a  codex       Codex Pro (team)  activef4e0a97b  glm         GLM Coding Plan   active03bd6e5f  kimi        Kimi K2 free      active… 334 more providers - + $ omniroute combo list - - + + OmniRoute Combos  ● always-on     [priority      ] enabled  ○ cost-saver    [cost-optimized] enabled  ○ fusion-panel  [fusion        ] enabled  ○ context-relay [context-relay ] enabled… run: omniroute combo create diff --git a/docs/diagrams/comparison-table.svg b/docs/diagrams/comparison-table.svg index d73a1fb2550..053194678cc 100644 --- a/docs/diagrams/comparison-table.svg +++ b/docs/diagrams/comparison-table.svg @@ -1,4 +1,4 @@ - + Static-header comparison table where each capability row fades in top to bottom; the OmniRoute column is highlighted and shows a check or a leading value in every row, while competitors show a mix of checks, partials and crosses. diff --git a/docs/diagrams/free-tier-budget.svg b/docs/diagrams/free-tier-budget.svg index e4503b5534e..393ee005941 100644 --- a/docs/diagrams/free-tier-budget.svg +++ b/docs/diagrams/free-tier-budget.svg @@ -1,4 +1,4 @@ - + @@ -63,7 +63,7 @@ ~1.51B FREE TOKENS / MONTH · STEADY up to ~2.13B in your first month — signup credits - documented free tiers · 41 provider pools · 495 models · one endpoint + documented free tiers · 40 provider pools · 495 models · one endpoint diff --git a/docs/diagrams/promise-pillars.svg b/docs/diagrams/promise-pillars.svg index 1c3f0a6cc9c..99b7f36b15a 100644 --- a/docs/diagrams/promise-pillars.svg +++ b/docs/diagrams/promise-pillars.svg @@ -1,4 +1,4 @@ - + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. @@ -21,7 +21,7 @@ - One endpoint. 346 providers. Never stop building — OmniRoute picks the cheapest one that works. + One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. @@ -38,7 +38,7 @@ Never hit limits - Auto-fallback across 346 providers in + Auto-fallback across 348 providers in milliseconds. Quota out? The next provider takes over — zero downtime. @@ -125,7 +125,7 @@ Production-grade - Circuit breakers, TLS stealth, MCP (109 + Circuit breakers, TLS stealth, MCP (110 tools), A2A, memory, guardrails, evals — 25,000+ tests. diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index fb332a45744..99543d2471e 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -28,7 +28,7 @@ Never stop coding. - Every AI tool → 346 providers — 90+ free — through one endpoint. + Every AI tool → 348 providers — 90+ free — through one endpoint. Claude Code · Codex · Cursor · Cline · Copilot · Antigravity  →  FREE Claude / GPT / Gemini · auto-fallback @@ -46,7 +46,7 @@ RTK + CAVEMAN · STACKED COMPRESSION - Save 15–95% tokens + Save 15–95% tokens diff --git a/docs/frameworks/MCP-SERVER.md b/docs/frameworks/MCP-SERVER.md index 18877057fd9..31be1172215 100644 --- a/docs/frameworks/MCP-SERVER.md +++ b/docs/frameworks/MCP-SERVER.md @@ -6,9 +6,9 @@ lastUpdated: 2026-08-08 # OmniRoute MCP Server Documentation -> Model Context Protocol server with 109 tools across routing, cache, compression, memory, skills, proxy, pool, Radar, and context source operations. +> Model Context Protocol server with 110 tools across routing, cache, compression, memory, skills, proxy, pool, Radar, and context source operations. > -> Source of truth: `open-sse/mcp-server/server.ts` computes **109 unique tools** with `countUniqueMcpTools()`: 44 canonical definitions (including the six CCR lifecycle tools, the agent-skills trio, and `omniroute_radar_catalog`), plus memory (3), skills (4), GitHub skills (3), pool (6), gamification (8), plugins (8), Notion (6), Obsidian (22), local corpus (3), and two RTK-only compression tools. +> Source of truth: `open-sse/mcp-server/server.ts` computes **110 unique tools** with `countUniqueMcpTools()`: 45 canonical definitions (including the six CCR lifecycle tools, the agent-skills trio, `omniroute_radar_catalog`, and `omniroute_x_search`), plus memory (3), skills (4), GitHub skills (3), pool (6), gamification (8), plugins (8), Notion (6), Obsidian (22), local corpus (3), and two RTK-only compression tools. ## Installation @@ -79,7 +79,8 @@ Cursor, Cline, and compatible MCP client setup. | `omniroute_list_models_catalog` | `read:models` | Full model catalog with capabilities, status, pricing | | `omniroute_radar_catalog` | `read:radar` | Local signed Radar catalog; optional provider/family filters | | `omniroute_tool_search` | `read:tools` | Discover tools from the registered MCP catalog | -| `omniroute_web_search` | `execute:search` | Web search through the configured search providers | +| `omniroute_web_search` | `execute:search` | Web search through the configured search providers. Not X/Twitter. | +| `omniroute_x_search` | `execute:search` | Search X (Twitter) through SuperGrok / xAI server-side `x_search`. Requires `xai-oauth` or an xAI API key. Not the X Developer Platform MCP. | | `omniroute_web_fetch` | `execute:search` | Fetch web content through the configured fetch providers | ## Advanced Tools (11) — Phase 2 @@ -226,7 +227,7 @@ See [AGENT-SKILLS.md](./AGENT-SKILLS.md) for the full catalog and how external a ## Related Frameworks (v3.8.0) -The MCP tool inventory above (109 unique tools, computed by `countUniqueMcpTools()`) is intentionally +The MCP tool inventory above (110 unique tools, computed by `countUniqueMcpTools()`) is intentionally scoped to runtime routing/cache/compression/memory/skills/proxy/context-source operations. Two adjacent frameworks ship alongside the MCP server in v3.8.0 and are documented separately: @@ -370,7 +371,7 @@ MCP tool, prompt, and resource registries can compress descriptions at registrat Description compression shrinks each tool's metadata; **tool-cardinality reduction** goes one step further by reducing _how many_ tools are announced at all. Advertising fewer tools in the `tools/list` manifest cuts the per-request token cost the client's model pays for the tool catalog ("layer 5" compression). The implementation is a pure, stateless filter in `open-sse/mcp-server/toolCardinality.ts` (`reduceToolManifest`), wired into the registration loop in `createMcpServer()` (`open-sse/mcp-server/server.ts`). -**Opt-in, off by default.** The filter only runs when at least one of two environment variables is set; with neither set, all 109 tools are announced unchanged. +**Opt-in, off by default.** The filter only runs when at least one of two environment variables is set; with neither set, all 110 tools are announced unchanged. | Variable | Mode | | :--------------- | :-------------------------------------------------------------------------------------- | diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 40487ee6cc1..b24e7b7f610 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -260,7 +260,28 @@ Memory behavior in Docker: - The image sets `OMNIROUTE_MEMORY_MB=1024` and derives `NODE_OPTIONS=--max-old-space-size=1024` from it. - The actual server process is started by the standalone launcher, which reads `OMNIROUTE_MEMORY_MB` and appends `--max-old-space-size=`. - Node uses the last repeated `--max-old-space-size` value, so setting `OMNIROUTE_MEMORY_MB` controls the effective Docker heap limit. -- Because the image always sets it, the launcher's own RAM-calibrated fallback never applies under Docker. Raise it explicitly (`-e OMNIROUTE_MEMORY_MB=2048`) on a host with headroom. +- Because the image always sets it, the launcher's own RAM-calibrated fallback never applies under Docker. Raise it explicitly for the workload (table below). `2048` is still too small for coding-agent `/v1/responses`. + +### Runtime RAM for coding agents + +The 1 GiB Docker default is a dashboard/light-chat floor, not a production size. Long `POST /v1/responses` bodies (hundreds of messages, tens of tools) retain multiple in-memory graphs during compression. Two overlapping ~3 MiB / ~750k-token requests have aborted V8 at a **12 GiB** old-space (`FATAL ERROR: Reached heap limit`) and also hit a 16 GiB cgroup OOM. See [#7849](https://github.com/diegosouzapw/OmniRoute/issues/7849). + +Size **cgroup `--memory` above the heap** — native buffers, SQLite, and compression intermediates sit outside V8. + +| Workload | `OMNIROUTE_MEMORY_MB` | Container / cgroup | Notes | +| --- | --- | --- | --- | +| Dashboard, one light chat | `1024` (image default) | ≥2 GiB | | +| One coding agent (Claude/Codex/Grok) | `8192` | ≥10 GiB | Typical single-session `/v1/responses` | +| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 GiB | Measured V8 abort at ~12 GiB heap | +| Three+ concurrent long contexts | do not on one process | serialize / more RAM | Default heavyweight admission is 1 in-flight; raising it without RAM reintroduces the abort | + +`omniroute serve` on bare metal calibrates ~35% of RAM (clamped `[512, 4096]`) when `OMNIROUTE_MEMORY_MB` is **unset**. Docker always sets `1024`, so that calibration never runs in the official image. + +```bash +docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ + -e OMNIROUTE_MEMORY_MB=8192 --memory=10g \ + -p 127.0.0.1:20128:20128 -v omniroute-data:/app/data diegosouzapw/omniroute:latest +``` ## Critical Environment Variables @@ -273,7 +294,7 @@ Beyond the defaults documented in [ENVIRONMENT.md](../reference/ENVIRONMENT.md), | `REDIS_PORT` | Host-side port for the bundled Redis container | `6379` | | `REDIS_BIND_HOST` | Host interface the bundled Redis port is published on (loopback unless you add AUTH) | `127.0.0.1` | | `AUTO_UPDATE_HOST_REPO_DIR` | Host path mounted into `cli` profile at `/workspace/omniroute` for self-update workflows | `.` (current directory) | -| `OMNIROUTE_MEMORY_MB` | Runtime Node heap ceiling for the Docker standalone server; overrides the image default above | `1024` | +| `OMNIROUTE_MEMORY_MB` | Runtime Node heap ceiling for the Docker standalone server; overrides the image default above. Coding agents: `8192`+ (see [runtime RAM](#runtime-ram-for-coding-agents)). | `1024` | | `DASHBOARD_PORT` / `API_PORT` | Override exposed ports for dashboard (20128) and API (20129) | `20128` / `20129` | | `OMNIROUTE_BASE_PATH` | URL subpath when the app is published behind a reverse proxy (e.g. `/omniroute`) | _(empty = root)_ | | `NEXT_PUBLIC_BASE_URL` | Public browser origin including the subpath (e.g. `https://host/omniroute`) | unset | @@ -484,7 +505,7 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. | Constraint | Consequence | | --- | --- | | Single writer | Do **not** run multiple replicas against the same SQLite file. That corrupts the DB. | -| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. | +| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. New requests during the empty-endpoint window get a reverse-proxy **`502 Bad Gateway: Unknown error`**, not OmniRoute JSON — clients cannot distinguish this from a provider failure (#11015). | | Same event loop as `/healthz` | A busy catalog or compression tick can delay probes; a short timeout then restarts the **only** replica. | **Probe matrix** (see also [Kubernetes probe recommendations](../ops/MONITORING_GUIDE.md#kubernetes-probe-recommendations)): @@ -497,7 +518,81 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. **Upgrades:** expect every session to drop. Drain clients if you can; there is no rolling update on default SQLite. Compose `restart: unless-stopped` plus Docker `HEALTHCHECK` will also replace the only process when the container is Unhealthy — same blast radius. -External Postgres / multi-writer HA is **not** a documented stock path. If you need HA, keep a single replica or run a topology the project has tested and documented separately. +Kubernetes snippet for a **single replica** (Recreate is required; do not raise `replicas` against one SQLite file): + +```yaml +spec: + replicas: 1 + strategy: + type: Recreate + template: + spec: + terminationGracePeriodSeconds: 90 + containers: + - name: omniroute + lifecycle: + preStop: + exec: + command: ["/bin/sleep", "15"] + readinessProbe: + httpGet: + path: /healthz + port: 20128 + periodSeconds: 5 + livenessProbe: + tcpSocket: + port: 20128 + periodSeconds: 20 +``` + +`preStop` sleep lets kube drop Service endpoints before SIGTERM so **new** traffic stops hitting the dying process. In-flight `/v1/responses` SSE is drained up to `SHUTDOWN_TIMEOUT_MS` (default 30s) via heavyweight admission leases (#11015). New requests that still reach the process get `503` + `Retry-After: 5`. The Recreate empty-endpoint gap until the replacement is Ready remains a hard outage — that is the SQLite topology, not a probe misconfig. + +External Postgres / multi-writer HA is **not** a documented stock path. If you need HA, keep a single replica or run a topology the project has tested and documented separately. The Postgres/MySQL work lives in [#8075](https://github.com/diegosouzapw/OmniRoute/issues/8075). Until that ships, the only supported way to multiply **large** `/v1/responses` capacity is N independent processes (next section), not `replicas > 1` on one volume. + +## Scale-out: N independent processes + +One Node process is **one V8 heap**. Two overlapping ~3 MiB / ~750k-token coding-agent `POST /v1/responses` (RTK + Caveman) abort that heap at ~12 Gi (`FATAL ERROR: Reached heap limit`) and can OOM a 16 Gi cgroup. See [#7849](https://github.com/diegosouzapw/OmniRoute/issues/7849). Raising `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` on that process reintroduces the abort. Small chats, `/healthz`, `/v1/models`, and MCP are **not** in that cap. + +To go beyond two concurrent **large** jobs **today**: + +| Do | Do not | +| --- | --- | +| Run **N containers/pods**, each with its **own** `DATA_DIR` / volume | Set `replicas > 1` against one SQLite file | +| Keep each instance at 1–2 heavy in-flight and 12–16 Gi cgroup | Give one process 8× RAM and `max=8` | +| Optional: `QUOTA_STORE_DRIVER=redis` + `QUOTA_STORE_REDIS_URL` for **shared quota counters** | Treat Redis as shared SQLite — it is not | +| Duplicate provider secrets into each instance (or accept partitioned dashboards) | Expect one dashboard / one call-log across instances | +| Front with any load balancer; sticky by API key or session is enough | Require a vendor-specific size-aware middleware | + +Hardware: `concurrent_large ≈ N × 2` at ~8–12 Gi heap / ~12–16 Gi cgroup **per instance**. Host RAM must cover `N × cgroup`, not “one 16 Gi pod with N=8.” + +Compose sketch (two heaps, two volumes — not `deploy.replicas: 2`): + +```yaml +services: + omniroute-a: + image: diegosouzapw/omniroute:3.8.49 + environment: + DATA_DIR: /app/data + OMNIROUTE_MEMORY_MB: "12288" + QUOTA_STORE_DRIVER: redis + QUOTA_STORE_REDIS_URL: redis://redis:6379 + volumes: [omniroute-a-data:/app/data] + ports: ["20128:20128"] + omniroute-b: + image: diegosouzapw/omniroute:3.8.49 + environment: + DATA_DIR: /app/data + OMNIROUTE_MEMORY_MB: "12288" + QUOTA_STORE_DRIVER: redis + QUOTA_STORE_REDIS_URL: redis://redis:6379 + volumes: [omniroute-b-data:/app/data] + ports: ["20138:20128"] +volumes: + omniroute-a-data: + omniroute-b-data: +``` + +In-process density (compression off the HTTP isolate) is [#11023](https://github.com/diegosouzapw/OmniRoute/issues/11023). One logical cluster on shared durable state is [#8075](https://github.com/diegosouzapw/OmniRoute/issues/8075). ## Important Notes diff --git a/docs/guides/ELECTRON_GUIDE.md b/docs/guides/ELECTRON_GUIDE.md index bdfcc247896..340e2a2c8df 100644 --- a/docs/guides/ELECTRON_GUIDE.md +++ b/docs/guides/ELECTRON_GUIDE.md @@ -252,7 +252,7 @@ AppImage signing is optional — set `LINUX_GPG_KEY` if signing. Artifacts land in `electron/dist-electron/`: -- `OmniRoute Setup X.Y.Z.exe`, `OmniRoute-X.Y.Z-portable.exe` (Windows) +- `OmniRoute.Setup.X.Y.Z.exe`, `OmniRoute X.Y.Z.exe` (Windows) - `OmniRoute-X.Y.Z-mac.dmg`, `OmniRoute-X.Y.Z-arm64-mac.dmg` (macOS) - `OmniRoute-X.Y.Z.AppImage`, `omniroute-desktop_X.Y.Z_amd64.deb` (Linux) diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index 0d237635cc1..8e0926137a5 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -263,6 +263,8 @@ Cost: currently listed as $0; terms and availability may change ### Cursor IDE +**Using Cursor as an OmniRoute client** (route Cursor chat through OmniRoute): + ``` Settings → Models → Advanced: OpenAI API Base URL: http://localhost:20128/v1 @@ -270,6 +272,10 @@ Settings → Models → Advanced: Model: cc/claude-opus-4-7 ``` +**Using OmniRoute as a Cursor provider** (OmniRoute calls Cursor upstream): prefer +**Dashboard → Providers → Cursor → Login with Cursor**. In Docker, see +[`docs/providers/CURSOR-DOCKER.md`](../providers/CURSOR-DOCKER.md). + ### Claude Code Edit `~/.claude/settings.json`: diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt index 53dfd1c0d6c..188e4c546fd 100644 --- a/docs/i18n/ar/llm.txt +++ b/docs/i18n/ar/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/az/llm.txt b/docs/i18n/az/llm.txt index cf0018a415e..1778044af16 100644 --- a/docs/i18n/az/llm.txt +++ b/docs/i18n/az/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt index cf0018a415e..1778044af16 100644 --- a/docs/i18n/bg/llm.txt +++ b/docs/i18n/bg/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/bn/llm.txt b/docs/i18n/bn/llm.txt index 2749b7b1081..e1f037d6698 100644 --- a/docs/i18n/bn/llm.txt +++ b/docs/i18n/bn/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt index 76a283564b7..c181079187f 100644 --- a/docs/i18n/cs/llm.txt +++ b/docs/i18n/cs/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt index 5bcb53ef104..fea53e17b2a 100644 --- a/docs/i18n/da/llm.txt +++ b/docs/i18n/da/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt index 17cb4a70bf6..494914668a9 100644 --- a/docs/i18n/de/llm.txt +++ b/docs/i18n/de/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt index 0441be519d1..6dfe5a95b36 100644 --- a/docs/i18n/es/llm.txt +++ b/docs/i18n/es/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/fa/llm.txt b/docs/i18n/fa/llm.txt index c81ec3a8534..55c6e845de3 100644 --- a/docs/i18n/fa/llm.txt +++ b/docs/i18n/fa/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt index 974084ef142..ca8eb2a75b3 100644 --- a/docs/i18n/fi/llm.txt +++ b/docs/i18n/fi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt index 6db495b3d95..6fc229f45d5 100644 --- a/docs/i18n/fr/llm.txt +++ b/docs/i18n/fr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/gu/llm.txt b/docs/i18n/gu/llm.txt index 885f78df633..0ef4d516e71 100644 --- a/docs/i18n/gu/llm.txt +++ b/docs/i18n/gu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt index 617ce45e8c7..e3d3b77ab8e 100644 --- a/docs/i18n/he/llm.txt +++ b/docs/i18n/he/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt index 67b0cfeba3d..f92757a9f3f 100644 --- a/docs/i18n/hi/llm.txt +++ b/docs/i18n/hi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt index fa74ab59976..e3ba1fa0a65 100644 --- a/docs/i18n/hu/llm.txt +++ b/docs/i18n/hu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt index 71572bd5a43..5a169f9e862 100644 --- a/docs/i18n/id/llm.txt +++ b/docs/i18n/id/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/in/llm.txt b/docs/i18n/in/llm.txt index c6f22ffc46c..e0b18cb0baa 100644 --- a/docs/i18n/in/llm.txt +++ b/docs/i18n/in/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt index db4f6a9cfda..dbde86f284b 100644 --- a/docs/i18n/it/llm.txt +++ b/docs/i18n/it/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt index f021f83d5a3..4bae4264258 100644 --- a/docs/i18n/ja/llm.txt +++ b/docs/i18n/ja/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt index 52cf209a74f..a2310374798 100644 --- a/docs/i18n/ko/llm.txt +++ b/docs/i18n/ko/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/mr/llm.txt b/docs/i18n/mr/llm.txt index cc3125f845f..d40eb377976 100644 --- a/docs/i18n/mr/llm.txt +++ b/docs/i18n/mr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt index 650ac31922e..cdf457c441e 100644 --- a/docs/i18n/ms/llm.txt +++ b/docs/i18n/ms/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt index bdd29b2d012..b2539aceb9e 100644 --- a/docs/i18n/nl/llm.txt +++ b/docs/i18n/nl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt index 7cb57aa53cd..9b4dedc1274 100644 --- a/docs/i18n/no/llm.txt +++ b/docs/i18n/no/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt index 73ea9e49ba4..d9b674c08eb 100644 --- a/docs/i18n/phi/llm.txt +++ b/docs/i18n/phi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md b/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md index 7bcf2bb0cfa..1b0495a1b01 100644 --- a/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md +++ b/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md @@ -252,7 +252,7 @@ Podpis AppImage jest opcjonalny — ustaw `LINUX_GPG_KEY`, jeśli podpisujesz. Artefakty lądują w `electron/dist-electron/`: -- `OmniRoute Setup X.Y.Z.exe`, `OmniRoute-X.Y.Z-portable.exe` (Windows) +- `OmniRoute.Setup.X.Y.Z.exe`, `OmniRoute X.Y.Z.exe` (Windows) - `OmniRoute-X.Y.Z-mac.dmg`, `OmniRoute-X.Y.Z-arm64-mac.dmg` (macOS) - `OmniRoute-X.Y.Z.AppImage`, `omniroute-desktop_X.Y.Z_amd64.deb` (Linux) diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt index 7d1d8be0f29..4c43c03d360 100644 --- a/docs/i18n/pl/llm.txt +++ b/docs/i18n/pl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt index 59be9758e4d..ecbeb9ad4c8 100644 --- a/docs/i18n/pt-BR/llm.txt +++ b/docs/i18n/pt-BR/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt index 1f4f205d44b..7cd77fdd4a6 100644 --- a/docs/i18n/pt/llm.txt +++ b/docs/i18n/pt/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt index 485f971d1f7..a4f03e94e7a 100644 --- a/docs/i18n/ro/llm.txt +++ b/docs/i18n/ro/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt index 319111231a5..8fe9b5bdb68 100644 --- a/docs/i18n/ru/llm.txt +++ b/docs/i18n/ru/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt index edb3477ef2f..c5a67d9242e 100644 --- a/docs/i18n/sk/llm.txt +++ b/docs/i18n/sk/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt index 48847120cb8..3b7b7b674ea 100644 --- a/docs/i18n/sv/llm.txt +++ b/docs/i18n/sv/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/sw/llm.txt b/docs/i18n/sw/llm.txt index f90e0ec031c..4c7bec4311b 100644 --- a/docs/i18n/sw/llm.txt +++ b/docs/i18n/sw/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ta/llm.txt b/docs/i18n/ta/llm.txt index 8f64c7aab8b..5323983d315 100644 --- a/docs/i18n/ta/llm.txt +++ b/docs/i18n/ta/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/te/llm.txt b/docs/i18n/te/llm.txt index 483d2f8a418..ce196f99275 100644 --- a/docs/i18n/te/llm.txt +++ b/docs/i18n/te/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt index ac3a14103aa..39eb9c04ff8 100644 --- a/docs/i18n/th/llm.txt +++ b/docs/i18n/th/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt index d6e2fbd674b..7f848b6cdd6 100644 --- a/docs/i18n/tr/llm.txt +++ b/docs/i18n/tr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt index 17e04f3ce16..bca18382ad2 100644 --- a/docs/i18n/uk-UA/llm.txt +++ b/docs/i18n/uk-UA/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/ur/llm.txt b/docs/i18n/ur/llm.txt index 9291f526424..85369bee2e9 100644 --- a/docs/i18n/ur/llm.txt +++ b/docs/i18n/ur/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt index 0cc3e4e22e3..da6b9b7179a 100644 --- a/docs/i18n/vi/llm.txt +++ b/docs/i18n/vi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt index 28d5e6fe234..0c737e5dc43 100644 --- a/docs/i18n/zh-CN/llm.txt +++ b/docs/i18n/zh-CN/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/i18n/zh-TW/llm.txt b/docs/i18n/zh-TW/llm.txt index ae8fd155b58..643943547df 100644 --- a/docs/i18n/zh-TW/llm.txt +++ b/docs/i18n/zh-TW/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -211,7 +211,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -266,7 +266,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -351,7 +351,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -438,7 +438,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -479,10 +479,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/docs/openapi.yaml b/docs/openapi.yaml index fb255d3543d..73941f37ead 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -2105,11 +2105,26 @@ paths: patch: tags: [Combos] summary: Update combo + description: >- + Partial update: the body is merged onto the stored combo, so a field left out keeps + its current value. An array that IS sent replaces the stored one outright. parameters: - $ref: "#/components/parameters/ResourceId" + requestBody: + required: true + content: + application/json: + schema: + type: object responses: "200": description: Updated combo + "400": + description: Invalid body, or the resulting combo fails validation + "404": + description: Combo not found + "409": + description: Name already taken, or the combo is quota-share managed delete: tags: [Combos] summary: Delete combo @@ -8878,12 +8893,20 @@ components: ComboCreate: type: object - required: [name, model] + required: [name, models] properties: name: type: string - model: - type: string + models: + type: array + minItems: 1 + items: + oneOf: + - type: string + description: "provider/model reference" + - type: object + description: "structured combo step (provider, model, weight, ...)" + additionalProperties: true strategy: type: string enum: @@ -8905,14 +8928,3 @@ components: - context-optimized - fusion default: priority - nodes: - type: array - items: - type: object - properties: - connectionId: - type: string - weight: - type: integer - priority: - type: integer diff --git a/docs/ops/REDIS_PRODUCTION_CONFIG.md b/docs/ops/REDIS_PRODUCTION_CONFIG.md index e8749c7bbcd..ada9ad44567 100644 --- a/docs/ops/REDIS_PRODUCTION_CONFIG.md +++ b/docs/ops/REDIS_PRODUCTION_CONFIG.md @@ -14,9 +14,12 @@ workloads: | Workload | Driver | Client Factory | Key Pattern | |---|---|---|---| -| Rate limiting | `rateLimiter.ts` | `getRedisClient()` — lazy `ioredis` singleton | Lua‑atomic rate limit windows | -| Auth cache | `apiKeys.ts` | Reuses `rateLimiter`'s client | `auth:api_key:` with TTL | -| Quota store | `redisQuotaStore.ts` | Separate `getRedisClient(url)` singleton | Configurable per-instance | +| Rate limiting | `rateLimiter.ts` | `getRedisClient()` — lazy `ioredis` singleton | `rl:*` Lua‑atomic rate limit windows | +| Auth cache | `apiKeys.ts` | Reuses `rateLimiter`'s client | `auth:api_key:` with TTL | +| Quota store | `redisQuotaStore.ts` | Separate `getRedisClient(url)` singleton | `quota:*` configurable per-instance | + +All three workloads share one namespace prefix so OmniRoute can co-exist with other apps on a +single Redis instance (e.g. `127.0.0.1:6379`). See [Key Namespacing](#key-namespacing). --- @@ -25,6 +28,7 @@ workloads: | Setting | Value | Where | |---|---|---| | `REDIS_URL` env var | `redis://redis:6379` (compose), optional | `rateLimiter.ts:5`, `.env.example` | +| `REDIS_KEY_PREFIX` env var | `omniroute:` (default) | `rateLimiter.ts`, `redisQuotaStore.ts`, `.env.example` | | `QUOTA_STORE_REDIS_URL` env var | separate, can differ from `REDIS_URL` | `quota/storeFactory.ts` | | `QUOTA_STORE_DRIVER` | `"sqlite"` (default), `"redis"` optional | `quota/storeFactory.ts` | | ioredis `maxRetriesPerRequest` | `3` | `rateLimiter.ts` client creation | @@ -36,6 +40,29 @@ workloads: --- +## Key Namespacing + +OmniRoute shares a Redis instance with whatever else runs on the host. Without a namespace, +keys like `auth:api_key:` or `rl:*` could collide with keys from other applications +using the same Redis (this instance runs Redis on `127.0.0.1:6379` alongside other services). + +Set `REDIS_KEY_PREFIX` to a non-empty string to prefix **every** OmniRoute key: + +```bash +# .env — all OmniRoute keys become omniroute:rl:*, omniroute:auth:*, omniroute:quota:* +REDIS_KEY_PREFIX=omniroute: +``` + +- **Default:** `omniroute:` (applied when `REDIS_KEY_PREFIX` is unset or blank). +- **Applied to:** rate limiter + auth cache (shared `ioredis` client via `keyPrefix`) and the + quota store (`KEY_PREFIX = "${REDIS_KEY_PREFIX}quota"`). +- **Changing the prefix** when keys already exist in Redis orphans the old keys (they expire + via TTL / LRU). Safe to change; no migration needed. +- **ioredis `keyPrefix`** automatically prepends the prefix on writes **and** strips it on reads, + so application code never sees the prefix. + +--- + ## Recommended Production Tuning ### 1. Connection Pool / Client Options (ioredis `Redis` constructor) diff --git a/docs/providers/CURSOR-DOCKER.md b/docs/providers/CURSOR-DOCKER.md index eb5724b0c8c..d524a8c8328 100644 --- a/docs/providers/CURSOR-DOCKER.md +++ b/docs/providers/CURSOR-DOCKER.md @@ -1,15 +1,58 @@ --- -title: "Cursor model listing" +title: "Cursor Provider in Docker Environments" version: 3.8.50 -lastUpdated: 2026-08-09 +lastUpdated: 2026-08-17 --- -# Cursor model listing +# Cursor Provider in Docker Environments -## Live catalog is exclusive when synced +When OmniRoute runs inside Docker, the legacy **Import from Cursor IDE** / +`cursor-agent` flows fail because the container cannot see the host Cursor +install. Use **Login with Cursor** (deep-control PKCE) instead. + +## Why IDE / CLI Import Fails in Docker + +1. **Filesystem isolation** — Auto-import looks for Linux paths such as + `~/.config/Cursor/User/globalStorage/state.vscdb` _inside_ the container. + On Docker Desktop for macOS the host IDE DB is not mounted by default, and + the container OS is Linux even when the host is Darwin. +2. **No `cursor-agent` binary** — Official OmniRoute images do not ship + `cursor-agent`. Available Models previously shelled out to + `cursor-agent --list-models` and fell back to a static catalog. +3. **Wrong binary** — Do **not** bind-mount a macOS `cursor-agent` into a Linux + container. It will not execute. + +## Recommended: Login with Cursor + +1. Open **Dashboard → Providers → Cursor**. +2. Choose the **Login with Cursor** tab. +3. Click **Login with Cursor** — OmniRoute opens + `https://cursor.com/loginDeepControl?…` in your **host** browser. +4. Approve the login in the browser, then return to the dashboard. OmniRoute + polls `api2.cursor.sh/auth/poll` until tokens arrive. +5. OmniRoute stores **access + refresh** tokens and refreshes them via + `https://api2.cursor.sh/auth/exchange_user_api_key`. + +This path does not require Cursor IDE or `cursor-agent` inside the container. + +## Model discovery + +With a logged-in connection, **Available Models / Auto-Sync** prefers Cursor’s +HTTP `AiService/AvailableModels` catalog using the connection bearer token. +If that fails, OmniRoute still tries host `cursor-agent` (when present), then +the static registry seed. + +OmniRoute always exposes **`auto`** in the catalog (display “Auto”), plus +OpenCodex-style router modes **`auto-cost`**, **`auto-balance`**, and +**`auto-intelligence`**. On the wire these map to Cursor’s `default` model +(with an `optimization` ModelParameter for the three variants). Prefer +`cu/auto` when premium models are out of usage — Auto often still has budget. + +### Live catalog is exclusive when synced After a successful Cursor model sync (`cursor-agent --list-models` → persisted -synced catalog), the **dashboard**, **`/v1/models`**, and **Test All** list: +synced catalog, or the bearer-authenticated `AvailableModels` fetch above), the +**dashboard**, **`/v1/models`**, and **Test All** list: 1. Models returned by the live sync 2. Injected auto-router ids: `auto`, `auto-cost`, `auto-balance`, `auto-intelligence` @@ -24,10 +67,63 @@ Effort-suffixed ids (for example `claude-4.6-sonnet-high`) may still be `ModelParameter`. Exclusive listing intentionally hides those static variants from Test All so probes match what Cursor actually returns as available. -## Helpers +### Helpers - `providerUsesExclusiveSyncedListing("cursor"|"cu")` — `src/lib/providers/modelListingCapability.ts` - `mergeProviderModelListing` — dashboard merge - `ensureCursorAutoCatalogEntry` — auto* inject on discovery + listing - `shouldSuppressStaticModelForExclusiveListing` — `/v1/models` static loop + +## Provider Limits (quota) + +**Usage → Provider Limits** for Cursor uses Bearer APIs on `api2.cursor.sh` +(`GetCurrentPeriodUsage` → usage summary → auth/usage) after PKCE or token +import. The legacy cookie/`cursor.com` dashboard path remains a last fallback +for older IDE-imported sessions. + +Windows typically include **Total**, **Auto + Composer**, and **API**. If +limits look empty, re-run **Login with Cursor** or re-import tokens (IDE import +alone is no longer required). + +## Empty turns / out of usage + +When Cursor accepts a Run but returns no assistant text (common when premium +usage is exhausted), OmniRoute surfaces an actionable **429** (quota cues) or +**502** with guidance — not a bare “Provider returned empty content”. Streaming +failures such as `not_found: AI Model Not Found` (usage window exhausted) are +classified as **Cursor rate limit / usage exceeded** and keep that message +through the SSE pipeline (the shared empty-stream guard does not overwrite an +already-emitted error). Check Provider Limits, try model **`auto`**, or raise +Cursor plan limits. + +## Client version (headless) + +Without a local `cursor-agent` install, OmniRoute resolves +`x-cursor-client-version` via env `CURSOR_AGENT_CLI_VERSION`, then a disk-cached +scrape of the Cursor installer script, then a pinned build id. Override with +`CURSOR_AGENT_CLI_VERSION` when needed. + +## Fallback: Manual Token Import + +If you cannot complete browser login: + +1. On the host, extract tokens from Cursor’s `state.vscdb`: + + ```bash + sqlite3 "$HOME/Library/Application Support/Cursor/User/globalStorage/state.vscdb" \ + "SELECT key, value FROM ItemTable WHERE key IN ('cursorAuth/accessToken','cursorAuth/refreshToken','storage.serviceMachineId');" + ``` + +2. Open **Import token** in the Cursor auth modal. +3. Paste **Access Token** and, when available, **Refresh Token** (required for + automatic refresh). Machine ID is optional. + +Access-token-only imports still work but will expire without a refresh token — +re-import when chat returns authentication errors. + +## Related + +- Zed Docker guidance: [`docs/providers/ZED-DOCKER.md`](./ZED-DOCKER.md) +- OpenCodex Cursor login reference (external): + https://github.com/lidge-jun/opencodex/blob/main/src/oauth/cursor.ts diff --git a/docs/reference/API_REFERENCE.md b/docs/reference/API_REFERENCE.md index 359ae4750ec..1b71551e373 100644 --- a/docs/reference/API_REFERENCE.md +++ b/docs/reference/API_REFERENCE.md @@ -636,6 +636,49 @@ completion. --- +## Self-service usage (`/api/usage/om-usage`) + +Any API key can read **its own** usage and quotas — no management auth. This is the endpoint a +client (CLI, the OmniCopilot panel) uses to show a key holder their spend. + +```bash +# Text form (the historical contract — plain text for a terminal) +curl -H "Authorization: Bearer " \ + http://localhost:20128/api/usage/om-usage + +# Structured form — what a UI consumes +curl -H "Authorization: Bearer " \ + "http://localhost:20128/api/usage/om-usage?format=json" +``` + +The key must have **`allowUsageCommand`** enabled (off by default — the dashboard's API-key +manager toggles it per key). Without it the endpoint answers `403`. + +`?format=json` returns a discriminated shape so a caller never reads a data field off a +refusal. On success: + +```jsonc +{ + "allowed": true, + // present only when the key opted into per-key usage limits (daily/weekly USD): + "personal": { "dailySpentUsd": 1.25, "dailyLimitUsd": 5, "dailyResetAtIso": "…", "weeklySpentUsd": 8, "weeklyLimitUsd": 20, "weeklyResetAtIso": "…" /* … */ }, + // the selected provider quota snapshot, or null when nothing is cached yet: + "provider": { "connectionId": "…", "provider": "claude", "plan": "…", "quotas": { /* … */ } }, + // every connection's snapshot, so a UI can render several providers side by side: + "providers": [ { "connectionId": "…", "provider": "claude", /* … */ }, { "provider": "codex", /* … */ } ] +} +``` + +On refusal (`401` bad key / `403` not allowed) the same route returns +`{ "allowed": false, "error": { "message": "…" } }` — a present-but-empty `personal`/`provider` +(key allowed, nothing learned yet) is a different state from a refusal, and only the JSON form +distinguishes them. + +**Auth:** the caller's own Bearer API key, validated with `isValidApiKey` — this is *not* the +management surface (`/api/keys/…`), which stays behind `requireManagementAuth`. + +--- + ## Semantic Cache ```bash diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 401d7f6a305..ae06e620905 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -198,11 +198,11 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | `REQUIRE_API_KEY` | `false` | API middleware | When `true`, all `/v1/*` proxy requests must include a valid API key. | | `ALLOW_API_KEY_REVEAL` | `false` | `src/shared/constants/featureFlagDefinitions.ts` | Allows revealing full API key values in the Dashboard UI. Configurable from Dashboard Feature Flags; security risk on shared instances. | | `NO_LOG_API_KEY_IDS` | _(empty)_ | `src/lib/compliance/index.ts` | Comma-separated API key IDs that bypass request logging (GDPR compliance). | -| `DEFAULT_RATE_LIMIT_PER_DAY` | `1000` | `src/shared/utils/apiKeyPolicy.ts` | Fallback per-day request budget applied to API keys whose `rate_limits` column is null. Default (unset/empty/malformed) keeps the legacy 1000/day, 5000/week, 20000/month windows. Set explicitly to `0` to opt out (unlimited). Any positive integer N enables N/day, 5N/week, 20N/month. Zod-validated; invalid values log a warning and use the legacy default. | +| `DEFAULT_RATE_LIMIT_PER_DAY` | _(unset = unlimited)_ | `src/shared/utils/apiKeyPolicy.ts` | Fallback per-day request budget applied to API keys whose `rate_limits` column is null. Unset or empty: no implicit cap (#2289, #11017). `0` is the same (unlimited). Positive integer N enables N/day, 5N/week, 20N/month. Malformed non-empty values fall back to the legacy 1000/day, 5000/week, 20000/month windows. | | `MAX_BODY_SIZE_BYTES` | `10485760` (10 MB) | `src/shared/middleware/bodySizeGuard.ts` | Maximum allowed request body size. Rejects payloads exceeding this limit. | | `OMNIROUTE_CHAT_LARGE_BODY_BYTES` | `262144` (256 KB) | `src/shared/middleware/chatBodyAdmission.ts` | Actual request bodies at or above this threshold require an atomic process-local heavyweight admission lease before JSON parsing. | | `OMNIROUTE_CHAT_HARD_MAX_BODY_BYTES` | `52428800` (50 MB) | `src/shared/middleware/chatBodyAdmission.ts` | Chat-route hard cap enforced against bytes read during bounded ingestion, including requests with missing, invalid, or dishonest `Content-Length`; excess receives `413`. | -| `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` | `1` | `src/shared/middleware/chatBodyAdmission.ts` | Maximum heavyweight chat requests admitted concurrently in one process. When capacity is unavailable, OmniRoute returns retryable `503` with `Retry-After`. | +| `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` | `1` | `src/shared/middleware/chatBodyAdmission.ts` | Maximum heavyweight chat requests admitted concurrently in **one process** (one V8 heap). Overload is retryable `503` with `Retry-After`. Two overlapping ~750k-token `/v1/responses` already abort ~12 Gi heaps (#7849); do not raise this to “use the host.” Multiply capacity with **N independent `DATA_DIR`s** (#11024), not `replicas>1` on one SQLite file. | | `OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO` | `0.75` | `src/shared/middleware/chatBodyAdmission.ts` | Heap-pressure shed ratio (`heapUsed / heap_size_limit`) for the structural admission gate (#10183, #10268). A second concurrent heavyweight request past `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` is only shed with the retryable `503` when the heap is ALSO at or above this ratio; on a healthy heap it is admitted instead. | | `OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM` | `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` (default `1`) | `src/shared/middleware/chatBodyAdmission.ts` | Bounded extra capacity for the healthy-heap fast path above (#10437). Without this bound, every busy-but-healthy-heap request bypassed admission with no ceiling at all — a slow leak or a burst that never quite trips the heap-shed ratio could still pile up unlimited concurrent heavyweight work. Once this many concurrent leases are active through the healthy-heap path, further busy requests fall through to the SAME bounded-wait/shed path used under real heap pressure. `0` disables the bypass entirely. | | `OMNIROUTE_CHAT_HEAVY_MESSAGE_COUNT` | `200` | `src/shared/middleware/chatBodyAdmission.ts` | Message count that classifies a chat request as heavyweight even when its body is below the byte threshold. | @@ -218,6 +218,8 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | `OMNIROUTE_ALLOW_PRIVATE_PROVIDER_URLS` | `false` | `src/shared/network/outboundUrlGuard.ts` | Allow provider URLs pointing to private/local networks (localhost, 192.168.x.x, 10.x.x.x, etc.). **REQUIRED for self-hosted providers** (LM Studio, Ollama, vLLM, Llamafile, Triton, SearXNG). When `false`, the dashboard rejects validation of local URLs. | | `OMNIROUTE_ALLOW_LOCAL_PROVIDER_URLS` | `true` | `src/shared/network/outboundUrlGuard.ts` | Allow adding/validating providers on local/private addresses (127.0.0.1, localhost, LAN, private ranges) — scoped to the provider validation path. **Default `true`** (local-first); set `false` to enforce strict public-only blocking. Cloud-metadata endpoints (169.254.169.254, metadata.google.internal) stay blocked regardless. (#5066) | | `AUDIO_REMOTE_PROVIDER_NODES` | `false` | `src/app/api/v1/_shared/audioProviderNodes.ts` | Let the `/v1/audio/*` routes (transcriptions, speech, translations) use an OpenAI-compatible provider node hosted outside localhost. Off by default — routing audio to a remote host changes egress identity and must be an explicit operator decision. Loopback/private nodes (localhost, 127.0.0.1, 172.16-31.x) are always allowed and unaffected. (#3963) | +| `OMNIROUTE_OIDC_DISABLE_PASSWORD_LOGIN` | `false` | `src/app/api/auth/login/route.ts` | When OIDC is enabled, disable password login so users can only authenticate via OIDC Single Sign-On. The bare alias `OIDC_DISABLE_PASSWORD_LOGIN` is also accepted; the Dashboard Feature Flag of the same key takes precedence. (#10889) | +| `OIDC_DISABLE_PASSWORD_LOGIN` | `false` | `src/app/api/auth/login/route.ts` | Bare alias of `OMNIROUTE_OIDC_DISABLE_PASSWORD_LOGIN` (#10889). | ### Hardening Checklist @@ -294,6 +296,7 @@ OmniRoute provides a two-layer defense: request-side injection scanning and resp | `CLOUD_URL` | _(empty)_ | `src/lib/cloudSync.ts` | Cloud relay endpoint URL (premium feature). | | `CLOUD_SYNC_TIMEOUT_MS` | `12000` | `src/lib/cloudSync.ts` | HTTP timeout for cloud sync requests. | | `OMNIROUTE_BUILD_PROFILE` | `full` | Webpack build config | Build-time profile (set to `minimal` to physically exclude privileged modules from bundle). | +| `OMNIROUTE_STANDALONE_DIR` | _.build/ standalone output_ | `scripts/build/colocate-standalone.mjs` | Build-time override for the standalone output directory consumed by the post-build colocation step. Not a runtime setting. | | `OMNIROUTE_CLOUD_SYNC_SECRET` | _(empty)_ | `src/lib/cloudSync.ts` | Shared secret used to verify the HMAC-SHA256 signature of Cloud Sync responses. | | `OMNIROUTE_CLOUD_SYNC_SECRETS` | `false` | `src/lib/cloudSync.ts` | Set to `true` to allow the Cloud Sync endpoint to overwrite local credentials. Default is `false`. | | `OMNIROUTE_ZED_IMPORT_LEGACY_ONE_STEP` | `false` | `src/app/api/providers/zed/import/route.ts` | Set to `true` to fall back to the v3.8.5 one-step "import everything" behavior without user confirmation. | @@ -733,7 +736,7 @@ REQUEST_TIMEOUT_MS (global override) | `OMNIROUTE_AGENT_GOAL_POLICY_ENABLED` | `true` | Kill-switch for the `/goal` heuristic. Set `false`/`0`/`off` to fully disable detection — readiness timeouts and stream recovery are never elevated by request body/headers, mitigating client-controlled timeout amplification. | | `OMNIROUTE_AGENT_GOAL_READINESS_MAX_TIMEOUT_MS` | `600000` | Maximum first-event readiness window for detected `/goal` agent runs or requests forced with `x-omniroute-agent-goal`. | | `OMNIROUTE_AGENT_GOAL_STREAM_RECOVERY` | `true` | Enable early stream recovery automatically for detected `/goal` agent runs. Set `false`/`0`/`off` to disable the goal-specific opt-in. This can only ADD recovery on top of the operator default — it never overrides an explicit `STREAM_RECOVERY_ENABLED`/DB settings opt-out. | -| `OMNIROUTE_CODEX_DROP_NONSTANDARD_EVENTS` | _(off)_ | Strip non-standard `codex.*` SSE events (e.g. `codex.rate_limits`) that break the OpenAI SDK's `responses.stream()` with a 502. Set `true`/`1`/`yes` to enable. | +| `OMNIROUTE_CODEX_DROP_NONSTANDARD_EVENTS` | `true` | Strip non-standard `codex.*` SSE events (e.g. `codex.rate_limits`) that break the OpenAI SDK's `responses.stream()` with a 502. Default ON (#11014). Set `0`/`false`/`no`/`off` to forward them. | | `FETCH_HEADERS_TIMEOUT_MS` | = `FETCH_TIMEOUT_MS` | Time to receive response headers. | | `OMNIROUTE_DIRECT_HEADERS_TIMEOUT_MS` | `30000` (30s) | Maximum response-start wait (ms) for each direct no-proxy attempt. A timeout retries once on a fresh socket; set `0` to disable the bound and retain the previous behavior. | | `FETCH_BODY_TIMEOUT_MS` | = `FETCH_TIMEOUT_MS` | Time to receive the full response body. | @@ -766,6 +769,8 @@ REQUEST_TIMEOUT_MS (global override) | `OMNIROUTE_NOTION_TLS_GRACE_MS` | `10000` | JS-side grace added on top of the wire timeout when the native binding is wedged. | | `OMNIROUTE_BROWSER_POOL` | `on` | Shared Playwright browser pool for browser-backed web-cookie chat (`browserPool.ts`); set `off` to disable. | | `WEB_COOKIE_USE_BROWSER` | `0` | Opt a web-cookie chat request into the browser-backed path (`browserBackedChat.ts`); `1` to enable. | +| `KIMI_WEB_BASE_URL` | `https://www.kimi.ai` | Base URL for the Kimi Web (international kimi.ai Connect-RPC) executor (`kimi-web.ts`); override only for mirror/proxy endpoints. | +| `KIMI_WEB_CHAT_URL` | `/apiv2/kimi.gateway.chat.v1.ChatService/Chat` | Full chat endpoint for the Kimi Web executor (`kimi-web.ts`). | | `OMNIROUTE_LOGIN_BROWSER_PATH` | _(auto-detected)_ | Path to a system Chrome/Edge executable for the Adobe Firefly interactive browser sign-in (`adobeFireflyBrowserLogin.ts`); overrides per-OS auto-detection. | Combo target attempts inherit the resolved upstream request timeout (`FETCH_TIMEOUT_MS`, or @@ -849,7 +854,7 @@ The logging system writes to both stdout and rotated log files. All configuratio | Variable | Default | Description | | -------------------------- | ------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `OMNIROUTE_MEMORY_MB` | _auto_ | **Recommended** Docker/standalone V8 heap limit (MB). When unset, calibrated dynamically (~35% of system RAM, clamped to `[512, 4096]`); `512` is only the floor when total memory can't be read. On `run-standalone.mjs` (Docker CMD), an **explicit** value is appended as `--max-old-space-size` and **wins** over a conflicting NODE_OPTIONS heap flag (V8 last-flag). `omniroute serve` still prefers an existing NODE_OPTIONS heap (#5238). Do not set both to different numbers — the process logs a warn naming both values and the winner. | +| `OMNIROUTE_MEMORY_MB` | _auto_ (bare metal); **`1024` in the Docker image** | **Recommended** Docker/standalone V8 heap limit (MB). When unset, calibrated dynamically (~35% of system RAM, clamped to `[512, 4096]`); `512` is only the floor when total memory can't be read. On `run-standalone.mjs` (Docker CMD), an **explicit** value is appended as `--max-old-space-size` and **wins** over a conflicting NODE_OPTIONS heap flag (V8 last-flag). `omniroute serve` still prefers an existing NODE_OPTIONS heap (#5238). Do not set both to different numbers — the process logs a warn naming both values and the winner. **The official Docker image always sets `1024`, so calibration never runs there.** Coding-agent `/v1/responses` needs `8192`–`12288` plus cgroup headroom — see [Docker Guide — runtime RAM](../guides/DOCKER_GUIDE.md#runtime-ram-for-coding-agents). | | `PROMPT_CACHE_MAX_SIZE` | `50` | Max cached system prompt entries. | | `PROMPT_CACHE_MAX_BYTES` | `2097152` (2 MB) | Max total prompt cache size. | | `PROMPT_CACHE_TTL_MS` | `300000` (5 min) | Prompt cache entry TTL. | @@ -905,6 +910,8 @@ Embedding layer, vector store and reranking knobs for the persistent memory subs ### Low-RAM Docker Example +`128` is dashboard-only. Coding agents on this heap `FATAL ERROR` during long `/v1/responses`. Do not use this example as a Claude/Codex/Grok gateway. + ```bash OMNIROUTE_MEMORY_MB=128 PROMPT_CACHE_MAX_SIZE=20 @@ -1018,7 +1025,6 @@ desktop install. | `NEXT_PUBLIC_DENO_RELAY_DEFAULT_PROJECT` | `omniroute-deno-relay` | `src/app/(dashboard)/dashboard/settings/components/proxy/DenoRelayModal.tsx` | Default Deno Deploy app name suggested in the proxy-pool "Deploy Relay" modal. | | `NEXT_PUBLIC_DENO_RELAY_ENABLED` | `true` | `src/app/(dashboard)/dashboard/settings/components/proxy/ProxyPoolTab.tsx` | Set to `false` to hide the Deno Deploy relay option from the Proxy Pool tab. | | `SEARCH_CACHE_TTL_MS` | `300000` (5 min) | `open-sse/services/searchCache.ts` | TTL for search API (Perplexity, Brave, etc.) response caching. | -| `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` | `false` | `src/app/api/providers/route.ts` | Allow multiple simultaneous connections per OpenAI-compatible provider. | | `ENABLE_CC_COMPATIBLE_PROVIDER` | `false` | `src/shared/utils/featureFlags.ts` | Reveal the experimental CC-compatible provider UI for Claude Code-only relays. | | `NINEROUTER_HOST` | `127.0.0.1` | `open-sse/executors/ninerouter.ts` | Override the host where the embedded 9router instance listens. | | `NINEROUTER_PORT` | `20130` | `open-sse/executors/ninerouter.ts` | Override the port where the embedded 9router instance listens. | @@ -1173,7 +1179,7 @@ AUTH_COOKIE_SECURE=true REQUIRE_API_KEY=true NEXT_PUBLIC_BASE_URL=https://omniroute.example.com BASE_URL=http://localhost:20128 -OMNIROUTE_MEMORY_MB=512 +OMNIROUTE_MEMORY_MB=8192 CORS_ORIGIN=https://your-frontend.example.com ``` @@ -1337,6 +1343,7 @@ Provider quota endpoints, network tunnels (Tailscale, Ngrok, MITM debug proxy), | `OMNIROUTE_REDIS_BIND_HOST` | `127.0.0.1` | `bin/cli/commands/redis.mjs` | Host interface the 1-click Redis launcher publishes on. The launcher starts Redis WITHOUT a password, so binding `0.0.0.0` hands every host on your LAN an unauthenticated Redis — only widen this if you also set a password on the instance yourself. | | `REDIS_BIND_HOST` | `127.0.0.1` | `docker-compose.yml` | Host interface docker-compose publishes the Redis sidecar on (#9286). The compose Redis runs without `requirepass`; app containers reach it over the compose network (`redis:6379`) — the published port exists only for host-side tooling. `0.0.0.0` exposes an unauthenticated Redis to the whole LAN. | | `REDIS_PORT` | `6379` | `docker-compose.yml` | Host port for the compose Redis sidecar. | +| `REDIS_KEY_PREFIX` | `omniroute:` | `src/shared/utils/rateLimiter.ts` | Namespace prefix applied to every OmniRoute Redis key (rate limiter, auth cache, quota store). Prevents key collisions when the Redis instance is shared with other apps (#11042). | | `OMNIROUTE_INTERNAL_SERVICE_TOKEN` | _(unset — mechanism disabled)_ | `src/lib/api/internalServiceAuth.ts` | Shared secret for identity-preserving internal REST hops (#9260): OmniRoute components calling other local OmniRoute routes send it as `x-omniroute-internal-service-token` so the original caller identity is preserved. Compared with `timingSafeEqual`. | | `OMNIROUTE_INTERNAL_SERVICE_TOKEN_FILE` | _(unset)_ | `src/lib/api/internalServiceAuth.ts` | Secret-file variant of the internal service token: path to a file whose trimmed content is the token. Only consulted when the inline var is unset. | | `OPENROUTER_PROVIDER_STATS_ENABLED` | `true` | `src/lib/catalog/openrouterProviderStats.ts` | Enrich the dashboard providers list with OpenRouter weekly ranking stats (#9324). On by default; set `false` to skip the background fetch entirely (non-blocking, never fatal). | @@ -1423,6 +1430,7 @@ value below unset in production deployments. | `ELECTRON_SMOKE_DATA_DIR` | _(tmpdir)_ | `scripts/dev/smoke-electron-packaged.mjs` | Data directory for the Electron smoke run. | | `ELECTRON_SMOKE_KEEP_DATA` | `0` | `scripts/dev/smoke-electron-packaged.mjs` | Set `1` to preserve the smoke data directory after the run. | | `ELECTRON_SMOKE_STREAM_LOGS` | `0` | `scripts/dev/smoke-electron-packaged.mjs` | Set `1` to stream Electron logs to stdout during the run. | +| `ELECTRON_SMOKE_COLD_RESTART` | `0` | `scripts/dev/smoke-electron-packaged.mjs` | #7592: relaunch against the same data dir and assert the second launch selects the native SQLite driver. | | `CLI_DEVIN_BIN` | _(PATH lookup)_ | `open-sse/executors/devin-cli.ts` | Override the Devin CLI binary path. | ### Docs translation pipeline diff --git a/docs/reference/FEATURE_FLAGS.md b/docs/reference/FEATURE_FLAGS.md index 8b46db649b7..a7afb802e0e 100644 --- a/docs/reference/FEATURE_FLAGS.md +++ b/docs/reference/FEATURE_FLAGS.md @@ -46,7 +46,7 @@ A boolean flag is considered **enabled** when its effective value is `"true"`, ## Flag Catalog -38 flags across 6 categories. **Default** is the definition default — the value +37 flags across 6 categories. **Default** is the definition default — the value used when neither a DB override nor an environment variable is present. ### Security (7) @@ -76,13 +76,12 @@ used when neither a DB override nor an environment variable is present. | `OMNIROUTE_ALLOW_LOCAL_PROVIDER_URLS` | boolean | `true` | | Allow adding/validating providers on local/private addresses (127.0.0.1, localhost, LAN). On by default (local-first); disable for strict public-only blocking. Cloud-metadata stays blocked. | | `ENABLE_CC_COMPATIBLE_PROVIDER` | boolean | `false` | ✓ | Enable Claude Code compatible provider mode. | -### Policies (4) +### Policies (3) | Key | Type | Default | Restart | Description | | ----------------------------------------- | ------- | ---------- | ------- | ---------------------------------------------------------------------- | | `TOOL_POLICY_MODE` | enum | `disabled` | | Tool-use policy enforcement mode. Values: `disabled`, `warn`, `block`. | | `RATE_LIMIT_AUTO_ENABLE` | boolean | `false` | | Automatically enable rate limiting based on usage patterns. | -| `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` | boolean | `false` | ✓ | Allow multiple connections per compatibility node. | | `DISABLE_CONTEXT_WINDOW_CHECKS` | boolean | `false` | | Skip OmniRoute's local context-window / max-input-token check for direct single-model requests. Upstream limits still apply. | ### Runtime (11) diff --git a/docs/reference/FREE_TIERS.md b/docs/reference/FREE_TIERS.md index a0ff4719159..927dccf7201 100644 --- a/docs/reference/FREE_TIERS.md +++ b/docs/reference/FREE_TIERS.md @@ -120,7 +120,6 @@ A 50-agent web-research pass (official docs + last-7-days news, adversarially ve | `firecrawl` | caution | Cloud API ToS has no explicit personal-proxy prohibition found, but the open-source self-hosted version is AGPL-3.0 (re… | | `gemini` | caution | ToS explicitly states the free tier is for "developers building with Google AI models for professional or business purp… | | `groq` | caution | Services Agreement §6.3 prohibits reselling, sublicensing, or distributing API access; §3.2 bars reselling/leasing acco… | -| `hackclub` | caution | Service is explicitly scoped to Hack Club teen members building projects/learning; no public ToS found explicitly permi… | | `huggingchat` | caution | Hugging Face ToS does not explicitly ban personal self-hosted proxies, but supplemental terms (referenced but not fully… | | `huggingface` | caution | ToS grants a limited license to access/use the service; the document does not explicitly permit or forbid a single-user… | | `hyperbolic` | caution | ToS grants API access "solely for your own personal or internal business purposes" and explicitly prohibits licensing, … | @@ -222,7 +221,6 @@ A 50-agent web-research pass (official docs + last-7-days news, adversarially ve | `duckduckgo-web` | keyless | — | — | avoid | 6 | | `freemodel-dev` | keyless | — | — | unknown | 4 | | `friendliai` | keyless | — | — | avoid | 2 | -| `hackclub` | keyless | — | — | caution | 3 | | `iflytek` | keyless | — | — | avoid | 1 | | `inference-net` | keyless | — | — | caution | 3 | | `liquid` | keyless | — | — | unknown | 1 | @@ -280,7 +278,6 @@ A 50-agent web-research pass (official docs + last-7-days news, adversarially ve - **`gitlawb`** — The shipped freeNote "Free tier available" is effectively stale. The original free MiMo access was removed in May 2026; the only remaining "free" option is a temporary promotional model (Nemotron 3 U… - **`gitlawb-gmi`** — Partially still accurate — free tier exists but is now narrowed to a single model (Nemotron 3 Ultra) after MiMo free access was revoked in late May 2026. The shipped note "Free tier available" unders… - **`groq`** — The shipped freeNote "30 RPM / 14.4K RPD" is accurate only for llama-3.1-8b-instant. Most other models (including llama-3.3-70b-versatile) have a much lower 1K RPD cap. The note omits model-specific … -- **`hackclub`** — The "30+ models" count appears accurate and still matches. The core offering remains free for Hack Club members. No evidence of tightening — still "$0 ALWAYS FREE" per the homepage. The freeNote omit… - **`huggingchat`** — The shipped freeNote ("Free LLM chat — no subscription required. Rate limits apply.") is partially accurate but significantly understates the restrictions. The free tier now operates on a hard $0.10/… - **`huggingface`** — Significantly tightened. The shipped freeNote ("Free Inference API for thousands of models") implied unlimited/generous free access, but as of mid-2025 the free tier is capped at $0.10/month in recur… - **`hyperbolic`** — Our shipped freeNote says "$1-5 trial credits on signup" — the $1 trial credit portion is accurate, but the "$5" figure refers to the minimum deposit required to unlock GPU rental (not free credits g… diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index 30eac06ad56..8a0a3b5d99a 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -1,16 +1,16 @@ --- title: "Provider Reference" version: 3.8.50 -lastUpdated: 2026-08-20 +lastUpdated: 2026-08-22 --- # Provider Reference > **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. > Regenerate with: `npm run gen:provider-reference` -> **Last generated:** 2026-08-20 +> **Last generated:** 2026-08-22 -Total providers: **346**. See category breakdown below. +Total providers: **348**. See category breakdown below. ## Categories @@ -34,7 +34,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each --- -## No-auth Providers (no key required) (11) +## No-auth Providers (no key required) (12) | ID | Alias | Name | Tags | Website | Notes | Tool calling | |----|-------|------|------|---------|-------|--------------| @@ -47,6 +47,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — | | `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — | | `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — | +| `uncloseai` | `unc` | UncloseAI | No-auth | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | — | | `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — | | `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — | @@ -62,8 +63,8 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `clinepass` | `cp` | ClinePass | OAuth | [link](https://cline.bot/cline-pass) | ClinePass is Cline's $9.99/mo subscription bundling 10 open coding models. Sign in with your Cline account (same login as the Cline CLI/IDE), or paste a direct ClinePass API key (app.cline.bot → Settings → API Keys). A ClinePass subscription unlocks the cline-pass/* models. Reuses the Cline WorkOS OAuth flow. | | `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. | | `codex` | `cx` | OpenAI Codex | OAuth | — | — | -| `cursor` | `cu` | Cursor IDE | OAuth, image | — | Image via Agent CLI (`CURSOR_AGENT_BIN`); same seat as chat | -| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | +| `cursor` | `cu` | Cursor IDE | OAuth | — | — | +| `devin-cli` | `dv` | Devin CLI | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | | `devin-desktop` | — | Devin Desktop | OAuth | [link](https://devin.ai) | Paste an existing Devin API key from an authenticated Devin session. Key export availability and steps vary by Devin version and account. | | `ghe-copilot` | `ghe-copilot` | GitHub Enterprise Copilot | OAuth | — | Enter your GHE instance URL (e.g., https://ghe.company.com) in provider settings, then authenticate via device flow. | | `github` | `gh` | GitHub Copilot | OAuth | — | — | @@ -98,7 +99,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — | | `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated | | `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — | -| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | +| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://chat.minimax.io) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | | `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — | | `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — | | `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated | @@ -149,7 +150,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | | `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | | `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | -| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | ⚠️ **DEPRECATED.** api.blackbox.ai returns HTTP 404 on every path variant (sweep 2026-08-21); the public inference surface has moved to the gated enterprise.blackbox.ai/v1 endpoint. | | `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | | `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | | `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | @@ -166,7 +167,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — | | `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | | `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | -| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /provider/v1/chat/completions endpoint. | | `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | | `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | | `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. | @@ -192,9 +193,9 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | | `free-ai` | `free-ai` | Free.ai | API key, aggregator | [link](https://free.ai) | 30,000 tokens/day cover self-hosted models after email verification. Usage beyond the pool can bill at raw cost, and premium external models are paid. | | `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | +| `freebuff` | `freebuff` | Freebuff | API key | [link](https://freebuff.com) | Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester). | | `freeinference` | `freeinference` | FreeInference | API key, aggregator | [link](https://freeinference.org) | Free research access without a card; non-Harvard applicants require manual approval and no numeric quota is publicly guaranteed. | | `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | -| `freepik` | `fpk` | Freepik (Mystic) | API key, image | [link](https://freepik.com) | Get API key at freepik.com/developers (Mystic image endpoint) | | `freetheai` | `fta` | FreeTheAi | API key, aggregator | [link](https://freetheai.xyz) | Join the FreeTheAi Discord to get your free API key. | | `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | | `g4f-gemini` | `g4fgem` | g4f.space — Gemini | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | @@ -213,7 +214,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | | `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | | `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | -| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `hackclub` | `hc` | Hackclub AI | API key | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | | `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | | `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | | `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | @@ -243,6 +244,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | | `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | | `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | +| `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | | `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | | `meganova-ai` | `meganova-ai` | MegaNova AI | API key, aggregator | [link](https://meganova.ai) | Free signup without a card. Published Tier 1 per-model quotas total 550 requests/day; they are not a shared global pool, and paid overage can apply if enabled. | | `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | @@ -330,7 +332,6 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | | `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. | | `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | -| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | | `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. | | `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | | `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | @@ -366,8 +367,8 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | | `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | | `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | -| `mlx-gemma` | `mlx-gemma` | MLX Gemma 26B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11435. Requires `uv` and `mlx-lm` installed. Model: `mlx-community/gemma-4-26B-A4B-it-qat-q4_0-mlx-aligned` (~15.9GB peak memory). | -| `mlx-qwen` | `mlx-qwen` | MLX Qwen 3.8 27B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11436. Requires `uv` and `mlx-lm` installed. Model: `maglun/Qwen3.8-27B-MLX-Mixed-3.80bpw` (~13.1GB peak memory). | +| `mlx-gemma` | `mlx-gemma` | MLX Gemma 26B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11435. Requires uv and mlx-lm installed. Model: mlx-community/gemma-4-26B-A4B-it-qat-q4_0-mlx-aligned (~15.9GB peak memory). | +| `mlx-qwen` | `mlx-qwen` | MLX Qwen 3.8 27B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11436. Requires uv and mlx-lm installed. Model: maglun/Qwen3.8-27B-MLX-Mixed-3.80bpw (~13.1GB peak memory). | | `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. | | `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | | `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | @@ -375,7 +376,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | | `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | -## Search Providers (12) +## Search Providers (13) | ID | Alias | Name | Tags | Website | Notes | |----|-------|------|------|---------|-------| @@ -390,6 +391,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | | `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | | `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | +| `x-search` | `x_search` | X Search (Grok) | Search | [link](https://docs.x.ai/developers/tools/x-search) | SuperGrok OAuth (xai-oauth) or xAI API key. This is Grok X Search, not the X Developer MCP. | | `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/business/api/) | X-API-Key from the You.com platform dashboard | ## Audio-only Providers (12) @@ -434,7 +436,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each - Catalog: [`src/shared/constants/providers.ts`](../../src/shared/constants/providers.ts) - Registry (per-model details): [`open-sse/config/providerRegistry.ts`](../../open-sse/config/providerRegistry.ts) -- Executors: [`open-sse/executors/`](../../open-sse/executors/) (105 implementations) +- Executors: [`open-sse/executors/`](../../open-sse/executors/) (106 implementations) - Translators: [`open-sse/translator/`](../../open-sse/translator/) ## See Also diff --git a/docs/routing/REASONING_REPLAY.md b/docs/routing/REASONING_REPLAY.md index c83f1edc5f3..46ed738e81b 100644 --- a/docs/routing/REASONING_REPLAY.md +++ b/docs/routing/REASONING_REPLAY.md @@ -26,7 +26,8 @@ But typical clients (Cursor, Cline, Roo Code, OpenAI SDK) strip `reasoning_conte ``` Turn N (assistant generates): → response contains reasoning_content + tool_calls - → cacheReasoningFromAssistantMessage() writes (memory + DB), keyed by every tool_call.id + → if requiresReasoningReplay(provider, model): cacheReasoningFromAssistantMessage() + writes (memory + DB), keyed by every tool_call.id → forward response to client (which may or may not retain reasoning) Turn N+1 (client sends follow-up): @@ -157,6 +158,7 @@ The cache exposes two endpoints under `src/app/api/cache/reasoning/route.ts`. Bo - **Cleanup:** `cleanupReasoningCache()` purges expired memory entries and runs `DELETE FROM reasoning_cache WHERE expires_at <= unixepoch('now')`. Health-check workers call this periodically. - **Crash recovery:** After a restart, memory is empty but the DB still holds unexpired entries. The first lookup for a given `tool_call_id` is a DB hit; subsequent lookups are memory hits. - **No reasoning, no cache:** `cacheReasoningFromAssistantMessage` returns `0` when the assistant message has no `reasoning_content` / `reasoning` field, so non-thinking responses cost nothing. +- **Write is gated too:** both call sites in `chatCore.ts` (non-streaming and streaming) only call `cacheReasoningFromAssistantMessage()` when `requiresReasoningReplay(provider, model)` is `true` — the same predicate the read side checks. Installs that never touch a replay provider stop paying for the write, the index update, and the try/catch on every reasoning-bearing response. - **Non-strict providers:** When `requiresReasoningReplay` is `false` and the target format is OpenAI, the translator **strips** any `reasoning_content` field from outgoing messages — OpenAI Chat Completions does not accept it. ## See Also diff --git a/docs/routing/STRICT_ZERO_COST.md b/docs/routing/STRICT_ZERO_COST.md new file mode 100644 index 00000000000..50f15778b6d --- /dev/null +++ b/docs/routing/STRICT_ZERO_COST.md @@ -0,0 +1,146 @@ +--- +title: "STRICT_ZERO_COST" +version: 3.8.50 +lastUpdated: 2026-08-20 +--- + +# STRICT_ZERO_COST + +> Opt-in, off by default (`settings.freeAccessPolicy !== "strict"` leaves every `auto/*` +> candidate pool byte-identical). A stricter sibling of `hidePaidModels` +> (`open-sse/services/autoCombo/paidModelFilter.ts`, #6512) for operators who need a hard +> guarantee against ANY incremental monetary spend, not just "documented as free". + +## Why this exists, and why `hidePaidModels` alone isn't enough + +`hidePaidModels` answers "is this model classified free in `FREE_MODEL_BUDGETS` right now?" — +a point-in-time catalog fact, checked via `isFreeModel()`/`providerHasFreeModels()` +(`src/shared/utils/freeModels.ts`). It says nothing about two real risks: + +1. A `recurring-*`/`one-time-initial` free tier's allowance can be **exhausted** — the catalog + still lists the model as free, but the account behind it has no headroom left. +2. Exceeding a free tier is not always a hard stop. Some providers document explicitly that no + payment method can ever be attached ("no credit card required"); others don't say, and a + handful bill automatically past the free allowance. + +`hidePaidModels` cannot distinguish these — it was never meant to. STRICT_ZERO_COST adds exactly +these two checks, evaluated per candidate, **before** category/tier ranking and **before** +dispatch — never after a request has already gone out. + +## Candidate classification + +For every candidate in the pool (`open-sse/services/autoCombo/virtualFactory.ts::buildPreparedPool`, +right after `filterPaidOnlyCandidates`): + +1. **Not in `FREE_MODEL_BUDGETS` at all** → excluded. This covers genuinely paid models and any + provider/model OmniRoute hasn't classified yet — new candidates start excluded, not included. +2. **`freeType: "keyless"`** → passes immediately, **but only for a candidate that genuinely + arrived via the no-auth path** (`connectionId === SYNTHETIC_NOAUTH_CONNECTION_ID`, + `open-sse/services/autoCombo/resilienceCandidateFilter.ts`). No credential exists for that + candidate, so no request against it can ever be billed — no runtime check is needed or + possible. The same catalogued `keyless` provider/model reached through a **real** DB + connection (`connectionId` is an actual connection id, or the candidate carries + `allowedConnectionIds`) does **not** get this shortcut — `keyless` metadata describes the + no-auth path specifically, not the provider in general, and never authorizes a real, + credentialed account. Such a candidate falls through to check 3 like any other, where it is + excluded unless the catalog entry separately carries `hardStopGuaranteed: true` (real + `keyless` entries never do — the shortcut was their only path to safety). +3. **Any other `freeType`** (`recurring-daily`, `recurring-monthly`, `recurring-credit`, + `recurring-uncapped`, `one-time-initial`, and any future type this module doesn't + special-case) → passes only if **all** of the following hold: + - `hardStopGuaranteed: true` is set on the catalog entry (`FreeModelBudget.hardStopGuaranteed`, + `open-sse/config/freeModelCatalog.ts`) — a **curated, hand-set fact** about the provider's + own published terms (e.g. an explicit "no credit card required" claim), never derived from + `freeType` or from a live API response. Unset (`undefined`) and `false` are both treated as + "not guaranteed". + - A usage adapter exists for the provider in `USAGE_FETCHER_PROVIDERS` + (`open-sse/services/usage.ts`) — the same registry that already backs the quota dashboard and + `getUsageForProvider()`. No adapter → excluded, permanently, until one is added. + - The live, cached `FreeAccessState` for **the specific connection actually being + evaluated** is `status: "SAFE"`, was checked within + `settings.autoRefreshProviderQuotaInterval` (default 180s — the existing setting, not a new + number), and reports `remainingFreeAllowance` above a small safety margin. +4. **`freeType: "discontinued"`** → always excluded. + +## Connection safety (per-connection verification, never per-candidate) + +A candidate in the auto-combo pool is not always tied to one connection. A "logical" candidate +(`connectionId: null`) carries an `allowedConnectionIds` allowlist — one or more actual +provider connections/accounts any of which could serve the request — and the account actually +used is decided later, at dispatch time, by `open-sse/services/combo/autoStrategy.ts` +(intersecting `allowedConnectionIds` against its own connection-selection logic, ~line 315-331). + +STRICT_ZERO_COST verifies the free-access state of **each connection in that allowlist +individually** (`evaluateCandidateConnections()` in `strictZeroCostFilter.ts`) and rewrites +`allowedConnectionIds` down to exactly the subset that came back `SAFE` — never the full +original list, and never a single arbitrarily-chosen member. Concretely: + +- Account A `SAFE`, account B `UNKNOWN`/exhausted/billable → only A remains selectable. +- All accounts `UNKNOWN` → the candidate is dropped entirely (empty safe set). +- A single-connection candidate (`connectionId` set directly, no allowlist) that fails is + dropped outright, never returned with an empty `allowedConnectionIds`. + +Because `autoStrategy.ts` already enforces `allowedConnectionIds` as a hard allowlist before +selecting a connection to dispatch to, rewriting it to the verified-SAFE subset is sufficient to +guarantee the connection actually used at dispatch is always one this filter itself verified — +never a different, unverified account on the same candidate. See +`tests/unit/autoCombo/strict-zero-cost-connection-safety.test.ts` for the regression proof +(keyless-bypass cases A/B/C, multi-account cases 1-5). + +`discovered automatically`: a provider/model shipped tomorrow with the right metadata (in the +catalog, with a usage adapter, `hardStopGuaranteed: true`) is usable the moment OmniRoute knows +about it — no code change, no whitelist entry, nothing to edit in this module. One removed from +the catalog disappears the same way. See +`tests/unit/autoCombo/strict-zero-cost-autodiscovery.test.ts` for the regression proof (via +injectable fixtures, not by mutating the real catalog). + +## Quota caching (`open-sse/services/autoCombo/freeAccessQuota.ts`) + +Reuses `getUsageForProvider()` — no second quota system. A short, in-memory, +process-lifetime cache sits in front of it (TTL equal to the default +`autoRefreshProviderQuotaInterval`) so a Telegram-scale request rate never triggers a live +billing-API call per candidate per request. Reads are synchronous: a cache miss returns +`undefined` (→ excluded, fail-closed) and kicks off a background refresh for the _next_ read — +nothing in the candidate-pool build path ever awaits a network call. + +`invalidateFreeAccessState(provider, connectionId)` is called from +`src/sse/services/auth.ts::markAccountUnavailable()` the moment a connection fails for any +reason, so the very next pool build reads a clean cache miss instead of a stale `SAFE` entry — +no waiting out the TTL after a 402/403/quota-exhausted response. + +## ToS guard (independent of economic safety) + +`excludeTosAvoid` (default `false`) drops any candidate whose curated `tos` verdict +(`FreeModelBudget.tos`) is `"avoid"` — reuses the same field `hidePaidModels`'s sibling docs +(`docs/reference/FREE_TIERS.md`) already populate. Deliberately separate from +`freeAccessPolicy`: a candidate can be economically `SAFE` and still excluded here for +contractual reasons, or left in when this guard is off even with `freeAccessPolicy: "strict"` on. + +## What passes today + +Run `npx tsx scripts/ad-hoc/dry-run-strict-zero-cost.ts` against a live instance's +`GET /v1/auto-combo/{channel}/candidates` output for a real before/after — the script now reads +each candidate's real `connectionId`, so it also proves the connection-safety fix live, not just +in unit tests. As of 2026-08-20, only `freeType: "keyless"` candidates pass in practice (7 of 29 +live candidates on this instance: `opencode/big-pickle`, `opencode/deepseek-v4-flash-free`, and +5 `felo-web` models — all confirmed arriving with the genuine no-auth `connectionId`, never a +real connection) — no currently-catalogued `recurring-*` provider both has a usage adapter +registered in `USAGE_FETCHER_PROVIDERS` **and** `hardStopGuaranteed: true` declared (e.g. `groq` +has neither the adapter registered here nor is fetched offline in this dry run; `kiro` lacks +`hardStopGuaranteed`). This is not a bug: it's the honest state of two independently-curated +metadata sets that happen not to overlap yet, not a limitation of the filter itself. + +With `excludeTosAvoid: true` added on top of the same live pool, the count drops from 7 to 0 — +every one of the 7 surviving candidates is curated `tos: "avoid"` today (`felo-web`, `opencode`). +This is a real, expected trade-off of turning the ToS guard on, not a bug: the guard is +`false` by default for exactly this reason (see "ToS guard" above). + +## Enabling + +```json +PUT /api/settings +{ "freeAccessPolicy": "strict", "excludeTosAvoid": false } +``` + +Both new settings default to their pre-feature values (`"off"` / `false`) — enabling neither +changes any existing `auto/*` routing behavior. diff --git a/electron/package.json b/electron/package.json index 57789a43189..bdb8b205f5d 100644 --- a/electron/package.json +++ b/electron/package.json @@ -135,6 +135,7 @@ "category": "Utility" }, "nsis": { + "artifactName": "${productName}.Setup.${version}.${ext}", "oneClick": false, "allowToChangeInstallationDirectory": true, "createDesktopShortcut": true, diff --git a/llm.txt b/llm.txt index da0da5334eb..51feb87bc32 100644 --- a/llm.txt +++ b/llm.txt @@ -1,6 +1,6 @@ # OmniRoute -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 346 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (109 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -14,7 +14,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.0.0 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 157 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 159 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -165,7 +165,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (346), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -207,7 +207,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── moderations.ts # Content moderation │ │ ├── rerank.ts # Reranking API │ │ └── search.ts # Web search API -│ ├── mcp-server/ # Built-in MCP server (109 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ ├── mcp-server/ # Built-in MCP server (110 tools, 3 transports: stdio/SSE/streamable-HTTP) │ │ ├── server.ts # MCP server core (tool registration, scope enforcement) │ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) │ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) @@ -262,7 +262,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ ├── i18n/ # 43-language translated docs │ ├── architecture/ # ARCHITECTURE.md, CODEBASE_DOCUMENTATION.md, REPOSITORY_MAP.md, AUTHZ_GUIDE.md, RESILIENCE_GUIDE.md, QUALITY_GATES.md │ ├── reference/ # API_REFERENCE.md, PROVIDER_REFERENCE.md, CLI-TOOLS.md -│ ├── frameworks/ # MCP-SERVER.md (109 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md +│ ├── frameworks/ # MCP-SERVER.md (110 tools), A2A-SERVER.md, SKILLS.md, MEMORY.md, CLOUD_AGENT.md, EVALS.md, WEBHOOKS.md │ ├── routing/ # AUTO-COMBO.md (14-factor scoring), REASONING_REPLAY.md │ ├── security/ # GUARDRAILS.md, COMPLIANCE.md, STEALTH_GUIDE.md, PUBLIC_CREDS.md, ERROR_SANITIZATION.md │ ├── guides/ # USER_GUIDE.md, TROUBLESHOOTING.md, ELECTRON_GUIDE.md, I18N.md @@ -277,7 +277,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **346 AI providers** with automatic format translation +- **348 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -347,7 +347,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### MCP Server (109 Tools) -109 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, +110 tools across modules: **44 canonical** (health, combos, quotas, routing, cost, models, cache, diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool**, **notion**, **obsidian**, **localCorpus**, **gamification**, and **plugin** modules. Full per-tool inventory: `docs/frameworks/MCP-SERVER.md`. @@ -434,7 +434,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 157 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (120 domain-specific files, 159 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. @@ -475,10 +475,10 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **346-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown -- **MCP server expanded to 109 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) +- **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) - **Cloud Agents** (Codex Cloud, Devin, Jules), **Guardrails**, **Evals**, **Webhooks**, **Compliance** frameworks - **Embedded services** manager (install/start/stop bundled services from the dashboard) - **Prompt compression** (RTK + Caveman codecs) saving up to ~95% tokens on eligible traffic diff --git a/open-sse/config/constants.ts b/open-sse/config/constants.ts index c45cf542a66..d99da4725b0 100644 --- a/open-sse/config/constants.ts +++ b/open-sse/config/constants.ts @@ -171,6 +171,7 @@ export const HTTP_STATUS = { FORBIDDEN: 403, NOT_FOUND: 404, NOT_ACCEPTABLE: 406, + UNPROCESSABLE_ENTITY: 422, REQUEST_TIMEOUT: 408, GONE: 410, RATE_LIMITED: 429, @@ -263,11 +264,17 @@ export const PROVIDER_PROFILES = { circuitBreakerReset: envInt("OMNIROUTE_CIRCUIT_BREAKER_API_KEY_RESET_MS", 30000), // Provider-level circuit breaker (entire provider cooldown after repeated failures) providerFailureThreshold: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_FAILURE_THRESHOLD", 15), // Scaled for 500+ connections (was 5) - providerFailureWindowMs: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_FAILURE_WINDOW_MS", 1800000), // 30min window (was 20min) + providerFailureWindowMs: envInt( + "OMNIROUTE_PROVIDER_BREAKER_API_KEY_FAILURE_WINDOW_MS", + 1800000 + ), // 30min window (was 20min) providerCooldownMs: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_COOLDOWN_MS", 600000), // 10min cooldown when threshold reached degradationThreshold: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_DEGRADATION_THRESHOLD", 7), maxBackoffMultiplier: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_MAX_BACKOFF_MULTIPLIER", 4), - backoffEscalationCount: envInt("OMNIROUTE_PROVIDER_BREAKER_API_KEY_BACKOFF_ESCALATION_COUNT", 3), + backoffEscalationCount: envInt( + "OMNIROUTE_PROVIDER_BREAKER_API_KEY_BACKOFF_ESCALATION_COUNT", + 3 + ), }, // Local providers (localhost inference backends like Ollama, LM Studio, oMLX). // Not yet wired into getProviderProfile() — will be used when local provider_nodes @@ -348,6 +355,23 @@ export const STREAM_RECOVERY = { HOLDBACK_MS: 750, BUFFER_MAX_BYTES: 65536, EARLY_RETRY_MAX: 4, + /** + * Minimum character overlap `trimContinuationOverlap` must find between the + * already-emitted text and a mid-stream continuation for the continuation to be + * accepted as a real resume, rather than an unrelated restart the model produced after + * ignoring the assistant-prefill. + * + * This is a DOCUMENTED TRADE-OFF, not a solved distinction: a model that continues + * cleanly with fewer than this many echoed characters (a legitimate, even preferred, + * outcome — there was nothing to de-duplicate) is indistinguishable, from string data + * alone, from a model that silently restarted on an unrelated sentence. Both produce a + * low/zero overlap. Rejecting below this threshold trades some false-positive rejections + * of legitimate low-overlap continuations (bounded retry, then a clean close — no data + * loss beyond that retry) against not silently gluing two unrelated fragments into one + * corrupted, unrecoverable answer. It does not eliminate the residual false negative + * either (an accidental coincidence at or above this many characters is still accepted). + */ + MIN_CONTINUATION_OVERLAP_CHARS: 8, } as const; /** diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index 8236a4d5d8f..38c17abf2d9 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -108,8 +108,9 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "bytez", modelId: "meta-llama/Llama-3.3-70B-Instruct", displayName: "meta-llama/Llama-3.3-70B-Instruct", monthlyTokens: 0, creditTokens: 1000000, freeType: "recurring-credit", poolKey: "bytez", tos: "ambiguous" }, { provider: "bytez", modelId: "mistralai/Mistral-7B-Instruct-v0.3", displayName: "mistralai/Mistral-7B-Instruct-v0.3", monthlyTokens: 0, creditTokens: 1000000, freeType: "recurring-credit", poolKey: "bytez", tos: "ambiguous" }, { provider: "bytez", modelId: "Qwen/Qwen2.5-72B-Instruct", displayName: "Qwen/Qwen2.5-72B-Instruct", monthlyTokens: 0, creditTokens: 1000000, freeType: "recurring-credit", poolKey: "bytez", tos: "ambiguous" }, - { provider: "cerebras", modelId: "zai-glm-4.7", displayName: "GLM 4.7", monthlyTokens: 30000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "cerebras", tos: "caution" }, - { provider: "cerebras", modelId: "gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 30000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "cerebras", tos: "caution" }, + // hardStopGuaranteed: Cerebras pricing page states "Free Trial: 1M tokens/day... no credit card" (open-sse/services/../providers/apikey/inference-hosts.ts:74-84). + { provider: "cerebras", modelId: "zai-glm-4.7", displayName: "GLM 4.7", monthlyTokens: 30000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "cerebras", tos: "caution", hardStopGuaranteed: true }, + { provider: "cerebras", modelId: "gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 30000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "cerebras", tos: "caution", hardStopGuaranteed: true }, // #8717: drop dead Workers AI ids (400/403/410). Keep Neurons/day budget on fp8-fast. { provider: "cloudflare-ai", modelId: "@cf/mistral/mistral-7b-instruct-v0.2-lora", displayName: "Mistral 7B (🆓)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "cloudflare-ai", tos: "caution" }, { provider: "cloudflare-ai", modelId: "@cf/qwen/qwen2.5-coder-32b-instruct", displayName: "Qwen 2.5 Coder 32B (🆓)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "cloudflare-ai", tos: "caution" }, @@ -187,14 +188,12 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "glm-cn", modelId: "glm-4.5-flash", displayName: "GLM-4.5-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, { provider: "glm-cn", modelId: "glm-4.7-flash", displayName: "GLM-4.7-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, { provider: "glm-cn", modelId: "glm-signup-bonus", displayName: "Z.AI — 20M signup bonus", monthlyTokens: 0, creditTokens: 20000000, freeType: "one-time-initial", poolKey: "zhipu-signup", tos: "ok" }, - { provider: "groq", modelId: "meta-llama/llama-4-scout-17b-16e-instruct", displayName: "Llama 4 Scout", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution" }, - { provider: "groq", modelId: "llama-3.3-70b-versatile", displayName: "Llama 3.3 70B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution" }, - { provider: "groq", modelId: "openai/gpt-oss-120b", displayName: "GPT-OSS 120B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution" }, - { provider: "groq", modelId: "openai/gpt-oss-20b", displayName: "GPT-OSS 20B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution" }, - { provider: "groq", modelId: "qwen/qwen3-32b", displayName: "Qwen3 32B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution" }, - { provider: "hackclub", modelId: "meta-llama/llama-3.3-70b-instruct", displayName: "Llama 3.3 70B", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "hackclub", tos: "caution" }, - { provider: "hackclub", modelId: "mistralai/mistral-7b-instruct", displayName: "Mistral 7B", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "hackclub", tos: "caution" }, - { provider: "hackclub", modelId: "deepseek-ai/deepseek-coder-33b", displayName: "DeepSeek Coder 33B", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "hackclub", tos: "caution" }, + // hardStopGuaranteed: Groq pricing page states "Free tier: 30 RPM / 14.4K RPD — no credit card" (open-sse/services/../providers/apikey/frontier-labs.ts:71-81). + { provider: "groq", modelId: "meta-llama/llama-4-scout-17b-16e-instruct", displayName: "Llama 4 Scout", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "llama-3.3-70b-versatile", displayName: "Llama 3.3 70B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "openai/gpt-oss-120b", displayName: "GPT-OSS 120B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "openai/gpt-oss-20b", displayName: "GPT-OSS 20B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "qwen/qwen3-32b", displayName: "Qwen3 32B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, { provider: "huggingchat", modelId: "baidu/ERNIE-4.5-VL-424B-A47B-Base-PT", displayName: "ERNIE 4.5 VL 424B A47B Base PT", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, { provider: "huggingchat", modelId: "CohereLabs/c4ai-command-r7b-12-2024", displayName: "Command R7B 12-2024", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, { provider: "huggingchat", modelId: "CohereLabs/command-a-reasoning-08-2025", displayName: "Command A Reasoning 08-2025", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, diff --git a/open-sse/config/freeModelCatalog.ts b/open-sse/config/freeModelCatalog.ts index af8b9b0d969..103f3775f7c 100644 --- a/open-sse/config/freeModelCatalog.ts +++ b/open-sse/config/freeModelCatalog.ts @@ -26,6 +26,20 @@ export interface FreeModelBudget { * reports this per model as `mayTrainOnYourPrompts` on its public catalog. */ trainsOnPrompts?: boolean; + /** + * True only when the provider's own published terms document that exceeding + * the free allowance is a hard stop (request refused / rate-limited) and NOT + * automatic pay-as-you-go billing — e.g. an explicit "no credit card + * required" claim on the provider's pricing page. This is a curated fact + * about the upstream provider, not something derivable from `freeType` or + * from any live API response, so it must be set by hand per entry with the + * source of the claim in a comment. Leave unset (undefined) whenever this + * isn't independently documented — `undefined` and `false` are both treated + * as "not guaranteed" by `strictZeroCostFilter.ts`; never default to `true` + * to grow the catalog. See STRICT_ZERO_COST in + * `open-sse/services/autoCombo/strictZeroCostFilter.ts`. + */ + hardStopGuaranteed?: boolean; } export interface FreeModelTotals { @@ -80,7 +94,7 @@ function fmt(n: number): string { function dedupedSum( models: FreeModelBudget[], pick: (m: FreeModelBudget) => number, - include: (m: FreeModelBudget) => boolean, + include: (m: FreeModelBudget) => boolean ): number { const poolMax = new Map(); let loose = 0; @@ -100,30 +114,30 @@ export function computeFreeModelTotals(opts: { excludeTosAvoid?: boolean } = {}) const steadyRecurringTokens = dedupedSum( models, (m) => m.monthlyTokens, - (m) => RECURRING.has(m.freeType), + (m) => RECURRING.has(m.freeType) ); const recurringCredits = dedupedSum( models, (m) => m.creditTokens, - (m) => m.freeType === "recurring-credit", + (m) => m.freeType === "recurring-credit" ); const oneTimeCredits = dedupedSum( models, (m) => m.creditTokens, - (m) => m.freeType === "one-time-initial", + (m) => m.freeType === "one-time-initial" ); const steadyWithRecurringCreditsTokens = steadyRecurringTokens + recurringCredits; const firstMonthRealisticTokens = steadyWithRecurringCreditsTokens + oneTimeCredits; const poolCount = new Set( - models.filter((m) => RECURRING.has(m.freeType) && m.poolKey).map((m) => m.poolKey), + models.filter((m) => RECURRING.has(m.freeType) && m.poolKey).map((m) => m.poolKey) ).size; // Deposit-unlock boost: sum the FREE_TIER_BOOSTS whose pool still has a live // recurring model in the (optionally ToS-filtered) set. const livePools = new Set( - models.filter((m) => RECURRING.has(m.freeType) && m.poolKey).map((m) => m.poolKey), + models.filter((m) => RECURRING.has(m.freeType) && m.poolKey).map((m) => m.poolKey) ); const boostMonthlyTokens = Object.entries(FREE_TIER_BOOSTS) .filter(([pool]) => livePools.has(pool)) diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 9c1580e7aed..8668de2c636 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -19,17 +19,16 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ export const GLM_SHARED_MODELS = Object.freeze([ { - // GLM-5.3 (2026-08-14): one upstream id; effort is the reasoning_effort - // param (low|high|max, default max) — the -high/-low entries below are - // OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. - // Default context window not yet published by Z.ai; 1M mirrored from - // GLM-5.2 (same base model). https://z.ai/blog/glm-5.3 + // GLM-5.3 exposes low|high|max reasoning_effort (default max); -high/-low + // are OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. + // https://docs.z.ai/guides/llm/glm-5.3 id: "glm-5.3", name: "GLM 5.3", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], }, { id: "glm-5.3-high", @@ -38,6 +37,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.3-low", @@ -46,14 +46,19 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low"], }, { + // GLM-5.2 has two positive effective tiers: low/medium map to high and xhigh + // maps to max; disabling thinking remains the separate thinking toggle. + // https://docs.z.ai/guides/capabilities/thinking id: "glm-5.2", name: "GLM 5.2", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high", "max"], }, { id: "glm-5.2-high", @@ -62,6 +67,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.2-max", @@ -70,14 +76,18 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["max"], }, { + // Earlier GLM families support the thinking toggle, not reasoning_effort. + // An explicit empty list prevents generic catalog tiers from being inferred. id: "glm-5.1", name: "GLM 5.1", contextLength: 204800, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5", @@ -86,6 +96,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5-turbo", @@ -94,6 +105,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7-flash", @@ -102,6 +114,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7", @@ -110,6 +123,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.6v", @@ -118,6 +132,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -127,6 +142,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5v", @@ -135,6 +151,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -144,6 +161,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5-air", @@ -152,6 +170,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, ]); diff --git a/open-sse/config/opencodeZenGoSharedModels.ts b/open-sse/config/opencodeZenGoSharedModels.ts new file mode 100644 index 00000000000..9f9d6628e91 --- /dev/null +++ b/open-sse/config/opencodeZenGoSharedModels.ts @@ -0,0 +1,16 @@ +/** + * Models declared identically in both the `opencode-zen` and `opencode-go` provider + * registries (same upstream family, opencode.ai/zen/*). Mirrors the GLM_SHARED_MODELS + * pattern in glmProvider.ts: one array, spread into each sibling RegistryEntry, so a + * metadata fix (targetFormat, supportsReasoning, ...) only has to land in one file + * instead of drifting out of sync across registries. + * + * Only entries that are byte-identical across both registries belong here — a model + * with tier-specific flags (e.g. go's effort variants, or a flag only one tier needs) + * stays local to that registry's own `models` array. + */ +export const OPENCODE_ZEN_GO_SHARED_MODELS = Object.freeze([ + { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, + { id: "qwen3.5-plus", name: "Qwen3.5 Plus", targetFormat: "claude", supportsVision: false }, + { id: "qwen3.6-plus", name: "Qwen3.6 Plus", targetFormat: "claude", supportsVision: false }, +]); diff --git a/open-sse/config/providerErrorRules.ts b/open-sse/config/providerErrorRules.ts index c910e3fb4d9..6f00d2c5f12 100644 --- a/open-sse/config/providerErrorRules.ts +++ b/open-sse/config/providerErrorRules.ts @@ -30,21 +30,63 @@ export type ProviderErrorRule = { export type ProviderErrorRuleMatch = { reason: ConfiguredErrorReason; /** - * Intended lock scope. #10334: this field is CONSUMED end-to-end only for - * providers in `HONORS_RULE_LOCK_SCOPE_PROVIDERS` (agentrouter-exclusive - * today, gated by `honorsRuleLockScope()`) — for those, `checkFallbackError` - * surfaces it as `ruleScope` on its return value for the persistence layer - * to honor instead of re-deriving scope from `hasPerModelQuota()`. For - * every other provider it remains INFORMATIONAL: `getProviderErrorRuleMatch` - * callers still read only `reason`/`cooldownMs`, and the actual lock scope - * is decided independently by each call site. Widening the allowlist is - * tracked as a follow-up — see `docs/architecture/RESILIENCE_GUIDE.md` §7. + * Intended lock scope. #10334: for a BUILT-IN catalog rule, this field is + * CONSUMED end-to-end only for providers in `HONORS_RULE_LOCK_SCOPE_PROVIDERS` + * (agentrouter-exclusive today, gated by `honorsRuleLockScope()`) — for those, + * `checkFallbackError` surfaces it as `ruleScope` on its return value for the + * persistence layer to honor instead of re-deriving scope from + * `hasPerModelQuota()`. For every other built-in-rule provider it remains + * INFORMATIONAL. #11104: an OPERATOR-declared rule (`OperatorProviderErrorRule`) + * is exempt from this allowlist — `honorsRuleLockScope()` always returns true + * when the provider has one, since the operator already opted in by declaring + * the rule. Widening `HONORS_RULE_LOCK_SCOPE_PROVIDERS` itself (for a new + * built-in catalog rule) is tracked as a follow-up — see + * `docs/architecture/RESILIENCE_GUIDE.md` §7. */ scope: "model" | "provider" | "connection"; /** Optional explicit cooldown; falls back to the existing per-reason defaults. */ cooldownMs?: number; }; +/** + * Operator-declared per-provider error rule (settings-driven). + * + * Mirrors the catalog `ProviderErrorRule` but is data-only so an operator can + * add a scope/cooldown/reason override for a provider without editing this + * file. `match` is a plain case-insensitive SUBSTRING of the error body — never + * a RegExp — so an operator-supplied pattern can never introduce a ReDoS on the + * error-classification hot path. Bounded to <= 50 rules total by the settings + * schema. An operator rule is consulted BEFORE the built-in `providerRuleRegistry` + * and wins on the first status+substring match for a provider. + */ +export type OperatorProviderErrorRule = { + status: number; + match: string; + scope: "model" | "provider" | "connection"; + reason?: ConfiguredErrorReason; + cooldownMs?: number; +}; + +let operatorProviderErrorRules: Record = {}; + +/** + * Inject operator-declared rules. Called from the runtime-settings applier + * (`applyRuntimeSettings`) once at boot and on every settings update, with the + * value validated by the settings schema. Pass `undefined`/empty/null to clear. + * Provider keys are lowercased so lookups are case-insensitive. + */ +export function setOperatorProviderErrorRules( + rules: Record | undefined | null +): void { + operatorProviderErrorRules = {}; + if (!rules) return; + for (const [provider, list] of Object.entries(rules)) { + if (Array.isArray(list) && list.length > 0) { + operatorProviderErrorRules[provider.toLowerCase()] = list; + } + } +} + // ─── Opencode ─────────────────────────────────────────────────────────────────── // Opencode Go uses an account-wide quota. The body usually says "rate limit // reached" but the presence of `x-ratelimit-remaining-requests: 0` is the @@ -272,11 +314,21 @@ export const providerRuleRegistry = new Map([ * FULL_TEXT_RULE_PROVIDERS: that set controls what body a rule matches against * (input), this one controls whether the matched scope changes caller behavior * (output). A provider could need one without the other. + * + * Providers with an operator-declared rule (`setOperatorProviderErrorRules`) + * are honored too, without being added here: the allowlist exists to gate + * BUILT-IN catalog rules, which change default behavior for every operator + * running that provider — an operator rule is already an explicit, per-operator + * opt-in, so gating it a second time behind this list would make the settings + * mechanism (#11104) silently inert for every provider except the ones listed + * below. See `hasOperatorRuleForProvider`. */ const HONORS_RULE_LOCK_SCOPE_PROVIDERS = new Set(["agentrouter"]); export function honorsRuleLockScope(provider: string | null | undefined): boolean { - return !!provider && HONORS_RULE_LOCK_SCOPE_PROVIDERS.has(provider.toLowerCase()); + if (!provider) return false; + const key = provider.toLowerCase(); + return HONORS_RULE_LOCK_SCOPE_PROVIDERS.has(key) || hasOperatorRuleForProvider(key); } /** @@ -310,28 +362,51 @@ export function egressBucketedLockProviders(): string[] { } /** - * Providers whose rules match on the FULL upstream error text. - * checkFallbackError's rule lookup normally passes only the structured + * Providers whose BUILT-IN catalog rules match on the FULL upstream error + * text. checkFallbackError's rule lookup normally passes only the structured * error ({code, type} — message stripped by the combo callers), which is * enough for header/status/code rules but blind to body-text markers like * agentrouter's "额度不足". Providers in this set get the raw error text as * the match body instead. EXCLUSIVE allowlist by owner decision (2026-08-13): * adding a provider here is an explicit opt-in — the default path for every * other provider must remain byte-for-byte unchanged. + * + * Operator-declared rules bypass this allowlist entirely (see + * `hasOperatorRuleForProvider`): the operator's `match` is a literal substring + * of the error body by construction, so a rule that never sees body text could + * never match anything, defeating the point of declaring it. */ const FULL_TEXT_RULE_PROVIDERS = new Set(["agentrouter"]); +/** + * True when an operator has declared at least one rule for this provider via + * `settings.providerErrorRules` (injected through `setOperatorProviderErrorRules`). + * Presence of the rule IS the opt-in — no separate allowlist to maintain, and + * no widening decision needed as new operators configure new providers. + */ +export function hasOperatorRuleForProvider(provider: string | null | undefined): boolean { + if (!provider) return false; + const rules = operatorProviderErrorRules[provider.toLowerCase()]; + return !!rules && rules.length > 0; +} + /** * Resolve the body handed to getProviderErrorRuleMatch inside - * checkFallbackError: full error text for FULL_TEXT_RULE_PROVIDERS, - * the structured error for everyone else. + * checkFallbackError: full error text for FULL_TEXT_RULE_PROVIDERS or any + * provider with an operator-declared rule, the structured error for everyone + * else. */ export function resolveRuleMatchBody( provider: string | null | undefined, structuredError: unknown, errorText: string | null | undefined ): unknown { - if (provider && FULL_TEXT_RULE_PROVIDERS.has(provider.toLowerCase()) && errorText) { + if ( + provider && + (FULL_TEXT_RULE_PROVIDERS.has(provider.toLowerCase()) || + hasOperatorRuleForProvider(provider)) && + errorText + ) { return errorText; } return structuredError ?? null; @@ -346,10 +421,32 @@ export function getProviderErrorRuleMatch( provider: string | null | undefined, status: number, headers: Headers | Record | null | undefined, - body?: unknown + body?: unknown, + operatorRules?: Record ): ProviderErrorRuleMatch | null { if (!provider) return null; - const rules = providerRuleRegistry.get(provider.toLowerCase()); + const key = provider.toLowerCase(); + + // Operator-declared rules win first: an operator can override any catalog + // rule for a provider without editing this file. `operatorRules` is the + // injected source (tests / direct callers); when omitted we fall back to the + // settings-backed cache populated by `setOperatorProviderErrorRules`. + const opRules = (operatorRules ?? operatorProviderErrorRules)?.[key]; + if (opRules && opRules.length > 0) { + const text = typeof body === "string" ? body : JSON.stringify(body ?? ""); + const lowered = text.toLowerCase(); + for (const r of opRules) { + if (r.status === status && lowered.includes(r.match.toLowerCase())) { + return { + reason: r.reason ?? "quota_exhausted", + scope: r.scope, + cooldownMs: r.cooldownMs, + }; + } + } + } + + const rules = providerRuleRegistry.get(key); if (!rules) return null; // Normalize headers: accept either a `Headers` object (from `fetch()`) or // a plain record. Provider rules access headers via plain object indexing. diff --git a/open-sse/config/providerFieldStrips.ts b/open-sse/config/providerFieldStrips.ts index 74282febdca..7568cd80bd3 100644 --- a/open-sse/config/providerFieldStrips.ts +++ b/open-sse/config/providerFieldStrips.ts @@ -56,12 +56,21 @@ export function stripGroqUnsupportedFields>(bo delete next.top_logprobs; if (Array.isArray(next.messages)) { next.messages = next.messages.map((m) => { - if (m && typeof m === "object" && "name" in m) { - const { name: _name, ...rest } = m as Record; + if (m && typeof m === "object") { + const { + name: _name, + model: _model, + messageId: _msgId, + sender: _sender, + ...rest + } = m as Record; + return rest; } + return m; }); } return next as T; } + diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts index bb3b6e0f9fe..69841c055ce 100644 --- a/open-sse/config/providerRegistry.ts +++ b/open-sse/config/providerRegistry.ts @@ -10,6 +10,10 @@ export { } from "./providers/registry/alibaba/index.ts"; export { REGISTRY } from "./providers/index.ts"; import { REGISTRY } from "./providers/index.ts"; +// Imported from `privateHost` rather than `outboundUrlGuard`: this module is reachable from +// `ProviderDetailPageClient.tsx`, so anything it pulls in has to survive a browser bundle +// (#11122). `privateHost` is platform-free by contract; the guard module is not. +import { isPrivateHost } from "@/shared/network/privateHost"; import { RegistryModel, REASONING_UNSUPPORTED, @@ -132,11 +136,8 @@ export function isLocalProvider(baseUrl?: string | null): boolean { try { const url = new URL(baseUrl); const hostname = url.hostname; - // Strictly matching 172.16.0.0/12 (Docker/local) and explicitly blocking ::1 per SSRF hardening - return ( - LOCAL_HOSTNAMES.has(hostname) || - /^172\.(1[6-9]|2[0-9]|3[0-1])\.\d{1,3}\.\d{1,3}$/.test(hostname) - ); + if (!hostname) return false; + return LOCAL_HOSTNAMES.has(hostname) || isPrivateHost(hostname); } catch { return false; } diff --git a/open-sse/config/providers/index.ts b/open-sse/config/providers/index.ts index 00861b24225..b0135fa5242 100644 --- a/open-sse/config/providers/index.ts +++ b/open-sse/config/providers/index.ts @@ -70,7 +70,6 @@ import { togetherProvider } from "./registry/together/index.ts"; import { cohereProvider } from "./registry/cohere/index.ts"; import { cursorProvider, cursor_apiProvider } from "./registry/cursor/index.ts"; import { volcengineProvider } from "./registry/volcengine/index.ts"; -import { hackclubProvider } from "./registry/hackclub/index.ts"; import { freetheaiProvider } from "./registry/freetheai/index.ts"; import { g4f_groqProvider } from "./registry/g4f-groq/index.ts"; import { g4f_geminiProvider } from "./registry/g4f-gemini/index.ts"; @@ -336,7 +335,6 @@ export const REGISTRY: Record = { cursor: cursorProvider, "cursor-api": cursor_apiProvider, volcengine: volcengineProvider, - hackclub: hackclubProvider, freetheai: freetheaiProvider, "g4f-groq": g4f_groqProvider, "g4f-gemini": g4f_geminiProvider, diff --git a/open-sse/config/providers/registry/blackbox/index.ts b/open-sse/config/providers/registry/blackbox/index.ts index 87d0ce26d72..46f58d35be6 100644 --- a/open-sse/config/providers/registry/blackbox/index.ts +++ b/open-sse/config/providers/registry/blackbox/index.ts @@ -5,6 +5,12 @@ export const blackboxProvider: RegistryEntry = { alias: "bb", format: "openai", executor: "default", + // NOTE: api.blackbox.ai returns HTTP 404 on /v1/chat/completions and /v1/models + // (empty body, all path variants) since sweep 2026-08-21; the public inference + // surface has moved to the gated enterprise.blackbox.ai/v1 endpoint. The provider + // is marked deprecated in src/shared/constants/providers/apikey/frontier-labs.ts — + // this registry entry is kept intact (registration/execution unaffected), so + // existing configured keys keep working if a restored/enterprise host is reachable. baseUrl: "https://api.blackbox.ai/v1/chat/completions", modelsUrl: "https://api.blackbox.ai/v1/models", authType: "apikey", diff --git a/open-sse/config/providers/registry/cline/index.ts b/open-sse/config/providers/registry/cline/index.ts index 21e8c600ed0..aecf811f193 100644 --- a/open-sse/config/providers/registry/cline/index.ts +++ b/open-sse/config/providers/registry/cline/index.ts @@ -27,7 +27,7 @@ export const clineProvider: RegistryEntry = { // the official free bucket and text-output models advertised as zero-cost. models: [ { - id: "zai/glm-5.2", + id: "z-ai/glm-5.2", name: "GLM 5.2", toolCalling: true, supportsReasoning: true, diff --git a/open-sse/config/providers/registry/command-code/index.ts b/open-sse/config/providers/registry/command-code/index.ts index 77b08ab8ce8..f298bcc21fb 100644 --- a/open-sse/config/providers/registry/command-code/index.ts +++ b/open-sse/config/providers/registry/command-code/index.ts @@ -8,7 +8,11 @@ export const command_codeProvider: RegistryEntry = { format: "openai", executor: "command-code", baseUrl: "https://api.commandcode.ai", - chatPath: "/alpha/generate", + // Chat uses the documented /provider/v1/chat/completions (OpenAI-format) + // endpoint — NOT the CLI-only /alpha/generate endpoint, which Command Code + // version-gates and proxy-blocks for external callers (#10265). Discovery + // already targets the sibling /provider/v1/models endpoint. + chatPath: "/provider/v1/chat/completions", modelsUrl: "https://api.commandcode.ai/provider/v1/models", // The discovery response is a partial routing catalog; static registry // entries omitted from it can still be accepted by the gateway. diff --git a/open-sse/config/providers/registry/cursor/index.ts b/open-sse/config/providers/registry/cursor/index.ts index 1bb6d02b214..67dfb744678 100644 --- a/open-sse/config/providers/registry/cursor/index.ts +++ b/open-sse/config/providers/registry/cursor/index.ts @@ -14,147 +14,228 @@ export const cursorProvider: RegistryEntry = { headers: getCursorRegistryHeaders(), clientVersion: CURSOR_REGISTRY_VERSION, models: [ - { id: "auto", name: "Auto (Server Picks)" }, - { id: "composer-2.5-fast", name: "Composer 2.5 Fast" }, - { id: "composer-2.5", name: "Composer 2.5" }, - { id: "composer-2-fast", name: "Composer 2 Fast" }, + { id: "auto", name: "Auto (current, default)" }, + { id: "auto-cost", name: "Auto (cost)" }, + { id: "auto-balance", name: "Auto (balance)" }, + { id: "auto-intelligence", name: "Auto (intelligence)" }, + // Legacy combo ids kept so existing cu/ targets are not orphaned. { id: "composer-2", name: "Composer 2" }, - // - { id: "gpt-5.5-none", name: "GPT 5.5 None" }, - { id: "gpt-5.5-none-fast", name: "GPT 5.5 None Fast" }, - { id: "gpt-5.5-low", name: "GPT 5.5 Low" }, - { id: "gpt-5.5-low-fast", name: "GPT 5.5 Low Fast" }, - { id: "gpt-5.5-medium", name: "GPT 5.5 Medium" }, - { id: "gpt-5.5-medium-fast", name: "GPT 5.5 Medium Fast" }, - { id: "gpt-5.5-high", name: "GPT 5.5 High" }, - { id: "gpt-5.5-high-fast", name: "GPT 5.5 High Fast" }, - { id: "gpt-5.5-extra-high", name: "GPT 5.5 Extra High" }, - { id: "gpt-5.5-extra-high-fast", name: "GPT 5.5 Extra High Fast" }, - // - { id: "gpt-5.4-low", name: "GPT 5.4 Low" }, + { id: "composer-2-fast", name: "Composer 2 Fast" }, { id: "gpt-5.4-low-fast", name: "GPT 5.4 Low Fast" }, - { id: "gpt-5.4-medium", name: "GPT 5.4 Medium" }, - { id: "gpt-5.4-medium-fast", name: "GPT 5.4 Medium Fast" }, - { id: "gpt-5.4-high", name: "GPT 5.4 High" }, - { id: "gpt-5.4-high-fast", name: "GPT 5.4 High Fast" }, - { id: "gpt-5.4-xhigh", name: "GPT 5.4 XHigh" }, - { id: "gpt-5.4-xhigh-fast", name: "GPT 5.4 XHigh Fast" }, - // - { id: "gpt-5.4-mini-none", name: "GPT 5.4 Mini None" }, - { id: "gpt-5.4-mini-low", name: "GPT 5.4 Mini Low" }, - { id: "gpt-5.4-mini-medium", name: "GPT 5.4 Mini Medium" }, - { id: "gpt-5.4-mini-high", name: "GPT 5.4 Mini High" }, - { id: "gpt-5.4-mini-xhigh", name: "GPT 5.4 Mini XHigh" }, - // - { id: "gpt-5.4-nano-none", name: "GPT 5.4 Nano None" }, - { id: "gpt-5.4-nano-low", name: "GPT 5.4 Nano Low" }, - { id: "gpt-5.4-nano-medium", name: "GPT 5.4 Nano Medium" }, - { id: "gpt-5.4-nano-high", name: "GPT 5.4 Nano High" }, - { id: "gpt-5.4-nano-xhigh", name: "GPT 5.4 Nano XHigh" }, - // { id: "gpt-5.3-codex-spark-preview-low", name: "GPT 5.3 Codex Spark Preview Low" }, { id: "gpt-5.3-codex-spark-preview", name: "GPT 5.3 Codex Spark Preview" }, { id: "gpt-5.3-codex-spark-preview-high", name: "GPT 5.3 Codex Spark Preview High" }, { id: "gpt-5.3-codex-spark-preview-xhigh", name: "GPT 5.3 Codex Spark Preview XHigh" }, - // - { id: "gpt-5.3-codex-low", name: "GPT 5.3 Codex Low" }, - { id: "gpt-5.3-codex-low-fast", name: "GPT 5.3 Codex Low Fast" }, - { id: "gpt-5.3-codex", name: "GPT 5.3 Codex" }, - { id: "gpt-5.3-codex-fast", name: "GPT 5.3 Codex Fast" }, - { id: "gpt-5.3-codex-high", name: "GPT 5.3 Codex High" }, - { id: "gpt-5.3-codex-high-fast", name: "GPT 5.3 Codex High Fast" }, - { id: "gpt-5.3-codex-xhigh", name: "GPT 5.3 Codex XHigh" }, - { id: "gpt-5.3-codex-xhigh-fast", name: "GPT 5.3 Codex XHigh Fast" }, - // - { id: "gpt-5.2-low", name: "GPT 5.2 Low" }, - { id: "gpt-5.2-low-fast", name: "GPT 5.2 Low Fast" }, - { id: "gpt-5.2", name: "GPT 5.2" }, - { id: "gpt-5.2-fast", name: "GPT 5.2 Fast" }, - { id: "gpt-5.2-high", name: "GPT 5.2 High" }, - { id: "gpt-5.2-high-fast", name: "GPT 5.2 High Fast" }, - { id: "gpt-5.2-xhigh", name: "GPT 5.2 XHigh" }, - { id: "gpt-5.2-xhigh-fast", name: "GPT 5.2 XHigh Fast" }, - // - { id: "claude-opus-4-8-low", name: "Claude Opus 4.8 Low" }, - { id: "claude-opus-4-8-low-fast", name: "Claude Opus 4.8 Low Fast" }, - { id: "claude-opus-4-8-medium", name: "Claude Opus 4.8 Medium" }, - { id: "claude-opus-4-8-medium-fast", name: "Claude Opus 4.8 Medium Fast" }, - { id: "claude-opus-4-8-high", name: "Claude Opus 4.8 High" }, - { id: "claude-opus-4-8-high-fast", name: "Claude Opus 4.8 High Fast" }, - { id: "claude-opus-4-8-xhigh", name: "Claude Opus 4.8 XHigh" }, - { id: "claude-opus-4-8-xhigh-fast", name: "Claude Opus 4.8 XHigh Fast" }, - { id: "claude-opus-4-8-max", name: "Claude Opus 4.8 Max" }, - { id: "claude-opus-4-8-max-fast", name: "Claude Opus 4.8 Max Fast" }, - { id: "claude-opus-4-8-thinking-low", name: "Claude Opus 4.8 Thinking Low" }, - { id: "claude-opus-4-8-thinking-low-fast", name: "Claude Opus 4.8 Thinking Low Fast" }, - { id: "claude-opus-4-8-thinking-medium", name: "Claude Opus 4.8 Thinking Medium" }, - { id: "claude-opus-4-8-thinking-medium-fast", name: "Claude Opus 4.8 Thinking Medium Fast" }, - { id: "claude-opus-4-8-thinking-high", name: "Claude Opus 4.8 Thinking High" }, - { id: "claude-opus-4-8-thinking-high-fast", name: "Claude Opus 4.8 Thinking High Fast" }, - { id: "claude-opus-4-8-thinking-xhigh", name: "Claude Opus 4.8 Thinking XHigh" }, - { id: "claude-opus-4-8-thinking-xhigh-fast", name: "Claude Opus 4.8 Thinking XHigh Fast" }, - { id: "claude-opus-4-8-thinking-max", name: "Claude Opus 4.8 Thinking Max" }, - { id: "claude-opus-4-8-thinking-max-fast", name: "Claude Opus 4.8 Thinking Max Fast" }, - // - { id: "claude-fable-5-low", name: "Claude Fable 5 Low" }, - { id: "claude-fable-5-medium", name: "Claude Fable 5 Medium" }, - { id: "claude-fable-5-high", name: "Claude Fable 5 High" }, - { id: "claude-fable-5-xhigh", name: "Claude Fable 5 XHigh" }, - { id: "claude-fable-5-max", name: "Claude Fable 5 Max" }, - { id: "claude-fable-5-thinking-low", name: "Claude Fable 5 Thinking Low" }, - { id: "claude-fable-5-thinking-medium", name: "Claude Fable 5 Thinking Medium" }, - { id: "claude-fable-5-thinking-high", name: "Claude Fable 5 Thinking High" }, - { id: "claude-fable-5-thinking-xhigh", name: "Claude Fable 5 Thinking XHigh" }, - { id: "claude-fable-5-thinking-max", name: "Claude Fable 5 Thinking Max" }, - // - { id: "claude-sonnet-5-low", name: "Claude Sonnet 5 Low" }, - { id: "claude-sonnet-5-medium", name: "Claude Sonnet 5 Medium" }, - { id: "claude-sonnet-5-high", name: "Claude Sonnet 5 High" }, - { id: "claude-sonnet-5-xhigh", name: "Claude Sonnet 5 XHigh" }, - { id: "claude-sonnet-5-max", name: "Claude Sonnet 5 Max" }, - { id: "claude-sonnet-5-thinking-low", name: "Claude Sonnet 5 Thinking Low" }, - { id: "claude-sonnet-5-thinking-medium", name: "Claude Sonnet 5 Thinking Medium" }, - { id: "claude-sonnet-5-thinking-high", name: "Claude Sonnet 5 Thinking High" }, - { id: "claude-sonnet-5-thinking-xhigh", name: "Claude Sonnet 5 Thinking XHigh" }, - { id: "claude-sonnet-5-thinking-max", name: "Claude Sonnet 5 Thinking Max" }, - // - { id: "claude-opus-4-7-low", name: "Claude Opus 4.7 Low" }, - { id: "claude-opus-4-7-medium", name: "Claude Opus 4.7 Medium" }, - { id: "claude-opus-4-7-high", name: "Claude Opus 4.7 High" }, - { id: "claude-opus-4-7-xhigh", name: "Claude Opus 4.7 XHigh" }, - { id: "claude-opus-4-7-max", name: "Claude Opus 4.7 Max" }, - - { id: "claude-opus-4-7-thinking-low", name: "Claude Opus 4.7 Thinking Low" }, - { id: "claude-opus-4-7-thinking-medium", name: "Claude Opus 4.7 Thinking Medium" }, - { id: "claude-opus-4-7-thinking-high", name: "Claude Opus 4.7 Thinking High" }, - { id: "claude-opus-4-7-thinking-xhigh", name: "Claude Opus 4.7 Thinking XHigh" }, - { id: "claude-opus-4-7-thinking-max", name: "Claude Opus 4.7 Thinking Max" }, - // - { id: "claude-4.6-opus-high", name: "Claude 4.6 Opus High" }, - { id: "claude-4.6-opus-high-thinking", name: "Claude 4.6 Opus High Thinking" }, { id: "claude-4.6-opus-high-thinking-fast", name: "Claude 4.6 Opus High Thinking Fast" }, - { id: "claude-4.6-opus-max", name: "Claude 4.6 Opus Max" }, - { id: "claude-4.6-opus-max-thinking", name: "Claude 4.6 Opus Max Thinking" }, { id: "claude-4.6-opus-max-thinking-fast", name: "Claude 4.6 Opus Max Thinking Fast" }, - // { id: "claude-4.6-sonnet-medium", name: "Claude 4.6 Sonnet Medium" }, { id: "claude-4.6-sonnet-medium-thinking", name: "Claude 4.6 Sonnet Medium Thinking" }, - // { id: "gemini-3.1-pro", name: "Gemini 3.1 Pro" }, - // { id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" }, { id: "gemini-3-flash", name: "Gemini 3 Flash" }, - // { id: "grok-4.6-medium", name: "Grok 4.6 Medium" }, { id: "grok-4.6-fast-medium", name: "Grok 4.6 Fast Medium" }, { id: "grok-4.6-high", name: "Grok 4.6 High" }, { id: "grok-4.6-fast-high", name: "Grok 4.6 Fast High" }, { id: "grok-4.6-xhigh", name: "Grok 4.6 XHigh" }, { id: "grok-4.6-fast-xhigh", name: "Grok 4.6 Fast XHigh" }, - // { id: "kimi-k3", name: "Kimi K3" }, { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, - ], + { id: "grok-4.3", name: "Grok 4.3" }, + { id: "grok-4.5-medium", name: "Grok 4.5 Medium" }, + { id: "grok-4.5-fast-medium", name: "Grok 4.5 Fast Medium" }, + { id: "grok-4.5-high", name: "Grok 4.5 High" }, + { id: "grok-4.5-fast-high", name: "Grok 4.5 Fast High" }, + { id: "grok-4.5-xhigh", name: "Grok 4.5 XHigh" }, + { id: "grok-4.5-fast-xhigh", name: "Grok 4.5 Fast XHigh" }, + { id: "kimi-k2.5", name: "Kimi K2.5" }, + { id: "gpt-5.3-codex-low", name: "Codex 5.3 Low" }, + { id: "gpt-5.3-codex-low-fast", name: "Codex 5.3 Low Fast" }, + { id: "gpt-5.3-codex", name: "Codex 5.3" }, + { id: "gpt-5.3-codex-fast", name: "Codex 5.3 Fast" }, + { id: "gpt-5.3-codex-high", name: "Codex 5.3 High" }, + { id: "gpt-5.3-codex-high-fast", name: "Codex 5.3 High Fast" }, + { id: "gpt-5.3-codex-xhigh", name: "Codex 5.3 Extra High" }, + { id: "gpt-5.3-codex-xhigh-fast", name: "Codex 5.3 Extra High Fast" }, + { id: "gpt-5.2", name: "GPT-5.2" }, + { id: "cursor-grok-4.5-high", name: "Cursor Grok 4.5" }, + { id: "cursor-grok-4.5-high-fast", name: "Cursor Grok 4.5 Fast" }, + { id: "composer-2.5", name: "Composer 2.5" }, + { id: "claude-opus-5-thinking-high", name: "Opus 5 1M Thinking" }, + { id: "claude-opus-5-thinking-high-fast", name: "Opus 5 1M Thinking Fast" }, + { id: "claude-opus-5-thinking-xhigh", name: "Opus 5 1M Extra High Thinking" }, + { id: "claude-opus-5-thinking-xhigh-fast", name: "Opus 5 1M Extra High Thinking Fast" }, + { id: "claude-opus-4-8-thinking-high", name: "Opus 4.8 1M Thinking" }, + { id: "claude-opus-4-8-thinking-high-fast", name: "Opus 4.8 1M Thinking Fast" }, + { id: "gpt-5.6-sol-high", name: "GPT-5.6 Sol 1M High" }, + { id: "gpt-5.6-sol-high-fast", name: "GPT-5.6 Sol High Fast" }, + { id: "gpt-5.6-sol-xhigh", name: "GPT-5.6 Sol 1M Extra High" }, + { id: "gpt-5.6-sol-xhigh-fast", name: "GPT-5.6 Sol Extra High Fast" }, + { id: "gpt-5.5-high", name: "GPT-5.5 1M High" }, + { id: "gpt-5.5-high-fast", name: "GPT-5.5 High Fast" }, + { id: "claude-fable-5-thinking-high", name: "Fable 5 1M Thinking (NO ZDR)" }, + { id: "claude-fable-5-thinking-xhigh", name: "Fable 5 1M Extra High Thinking (NO ZDR)" }, + { id: "claude-sonnet-5-thinking-high", name: "Sonnet 5 1M Thinking" }, + { id: "claude-sonnet-5-thinking-xhigh", name: "Sonnet 5 1M Extra High Thinking" }, + { id: "kimi-k3-high", name: "Kimi K3 High" }, + { id: "cursor-grok-4.5-low", name: "Cursor Grok 4.5 Low" }, + { id: "cursor-grok-4.5-low-fast", name: "Cursor Grok 4.5 Low Fast" }, + { id: "cursor-grok-4.5-medium", name: "Cursor Grok 4.5 Medium" }, + { id: "cursor-grok-4.5-medium-fast", name: "Cursor Grok 4.5 Medium Fast" }, + { id: "composer-2.5-fast", name: "Composer 2.5 Fast" }, + { id: "claude-opus-5-low", name: "Opus 5 1M Low" }, + { id: "claude-opus-5-low-fast", name: "Opus 5 1M Low Fast" }, + { id: "claude-opus-5-medium", name: "Opus 5 1M Medium" }, + { id: "claude-opus-5-medium-fast", name: "Opus 5 1M Medium Fast" }, + { id: "claude-opus-5-high", name: "Opus 5 1M" }, + { id: "claude-opus-5-high-fast", name: "Opus 5 1M Fast" }, + { id: "claude-opus-5-thinking-low", name: "Opus 5 1M Low Thinking" }, + { id: "claude-opus-5-thinking-low-fast", name: "Opus 5 1M Low Thinking Fast" }, + { id: "claude-opus-5-thinking-medium", name: "Opus 5 1M Medium Thinking" }, + { id: "claude-opus-5-thinking-medium-fast", name: "Opus 5 1M Medium Thinking Fast" }, + { id: "claude-opus-5-thinking-max", name: "Opus 5 1M Max Thinking" }, + { id: "claude-opus-5-thinking-max-fast", name: "Opus 5 1M Max Thinking Fast" }, + { id: "claude-opus-4-8-low", name: "Opus 4.8 1M Low" }, + { id: "claude-opus-4-8-low-fast", name: "Opus 4.8 1M Low Fast" }, + { id: "claude-opus-4-8-medium", name: "Opus 4.8 1M Medium" }, + { id: "claude-opus-4-8-medium-fast", name: "Opus 4.8 1M Medium Fast" }, + { id: "claude-opus-4-8-high", name: "Opus 4.8 1M" }, + { id: "claude-opus-4-8-high-fast", name: "Opus 4.8 1M Fast" }, + { id: "claude-opus-4-8-xhigh", name: "Opus 4.8 1M Extra High" }, + { id: "claude-opus-4-8-xhigh-fast", name: "Opus 4.8 1M Extra High Fast" }, + { id: "claude-opus-4-8-max", name: "Opus 4.8 1M Max" }, + { id: "claude-opus-4-8-max-fast", name: "Opus 4.8 1M Max Fast" }, + { id: "claude-opus-4-8-thinking-low", name: "Opus 4.8 1M Low Thinking" }, + { id: "claude-opus-4-8-thinking-low-fast", name: "Opus 4.8 1M Low Thinking Fast" }, + { id: "claude-opus-4-8-thinking-medium", name: "Opus 4.8 1M Medium Thinking" }, + { id: "claude-opus-4-8-thinking-medium-fast", name: "Opus 4.8 1M Medium Thinking Fast" }, + { id: "claude-opus-4-8-thinking-xhigh", name: "Opus 4.8 1M Extra High Thinking" }, + { id: "claude-opus-4-8-thinking-xhigh-fast", name: "Opus 4.8 1M Extra High Thinking Fast" }, + { id: "claude-opus-4-8-thinking-max", name: "Opus 4.8 1M Max Thinking" }, + { id: "claude-opus-4-8-thinking-max-fast", name: "Opus 4.8 1M Max Thinking Fast" }, + { id: "gpt-5.6-sol-none", name: "GPT-5.6 Sol 1M None" }, + { id: "gpt-5.6-sol-none-fast", name: "GPT-5.6 Sol None Fast" }, + { id: "gpt-5.6-sol-low", name: "GPT-5.6 Sol 1M Low" }, + { id: "gpt-5.6-sol-low-fast", name: "GPT-5.6 Sol Low Fast" }, + { id: "gpt-5.6-sol-medium", name: "GPT-5.6 Sol 1M" }, + { id: "gpt-5.6-sol-medium-fast", name: "GPT-5.6 Sol Fast" }, + { id: "gpt-5.6-sol-max", name: "GPT-5.6 Sol 1M Max" }, + { id: "gpt-5.6-sol-max-fast", name: "GPT-5.6 Sol Max Fast" }, + { id: "gpt-5.5-none", name: "GPT-5.5 1M None" }, + { id: "gpt-5.5-none-fast", name: "GPT-5.5 None Fast" }, + { id: "gpt-5.5-low", name: "GPT-5.5 1M Low" }, + { id: "gpt-5.5-low-fast", name: "GPT-5.5 Low Fast" }, + { id: "gpt-5.5-medium", name: "GPT-5.5 1M" }, + { id: "gpt-5.5-medium-fast", name: "GPT-5.5 Fast" }, + { id: "gpt-5.5-extra-high", name: "GPT-5.5 1M Extra High" }, + { id: "gpt-5.5-extra-high-fast", name: "GPT-5.5 Extra High Fast" }, + { id: "claude-fable-5-low", name: "Fable 5 1M Low (NO ZDR)" }, + { id: "claude-fable-5-medium", name: "Fable 5 1M Medium (NO ZDR)" }, + { id: "claude-fable-5-high", name: "Fable 5 1M (NO ZDR)" }, + { id: "claude-fable-5-xhigh", name: "Fable 5 1M Extra High (NO ZDR)" }, + { id: "claude-fable-5-max", name: "Fable 5 1M Max (NO ZDR)" }, + { id: "claude-fable-5-thinking-low", name: "Fable 5 1M Low Thinking (NO ZDR)" }, + { id: "claude-fable-5-thinking-medium", name: "Fable 5 1M Medium Thinking (NO ZDR)" }, + { id: "claude-fable-5-thinking-max", name: "Fable 5 1M Max Thinking (NO ZDR)" }, + { id: "claude-sonnet-5-low", name: "Sonnet 5 1M Low" }, + { id: "claude-sonnet-5-medium", name: "Sonnet 5 1M Medium" }, + { id: "claude-sonnet-5-high", name: "Sonnet 5 1M" }, + { id: "claude-sonnet-5-xhigh", name: "Sonnet 5 1M Extra High" }, + { id: "claude-sonnet-5-max", name: "Sonnet 5 1M Max" }, + { id: "claude-sonnet-5-thinking-low", name: "Sonnet 5 1M Low Thinking" }, + { id: "claude-sonnet-5-thinking-medium", name: "Sonnet 5 1M Medium Thinking" }, + { id: "claude-sonnet-5-thinking-max", name: "Sonnet 5 1M Max Thinking" }, + { id: "gpt-5.6-terra-none", name: "GPT-5.6 Terra 1M None" }, + { id: "gpt-5.6-terra-none-fast", name: "GPT-5.6 Terra None Fast" }, + { id: "gpt-5.6-terra-low", name: "GPT-5.6 Terra 1M Low" }, + { id: "gpt-5.6-terra-low-fast", name: "GPT-5.6 Terra Low Fast" }, + { id: "gpt-5.6-terra-medium", name: "GPT-5.6 Terra 1M" }, + { id: "gpt-5.6-terra-medium-fast", name: "GPT-5.6 Terra Fast" }, + { id: "gpt-5.6-terra-high", name: "GPT-5.6 Terra 1M High" }, + { id: "gpt-5.6-terra-high-fast", name: "GPT-5.6 Terra High Fast" }, + { id: "gpt-5.6-terra-xhigh", name: "GPT-5.6 Terra 1M Extra High" }, + { id: "gpt-5.6-terra-xhigh-fast", name: "GPT-5.6 Terra Extra High Fast" }, + { id: "gpt-5.6-terra-max", name: "GPT-5.6 Terra 1M Max" }, + { id: "gpt-5.6-terra-max-fast", name: "GPT-5.6 Terra Max Fast" }, + { id: "claude-opus-4-7-low", name: "Opus 4.7 1M Low" }, + { id: "claude-opus-4-7-low-fast", name: "Opus 4.7 1M Low Fast" }, + { id: "claude-opus-4-7-medium", name: "Opus 4.7 1M Medium" }, + { id: "claude-opus-4-7-medium-fast", name: "Opus 4.7 1M Medium Fast" }, + { id: "claude-opus-4-7-high", name: "Opus 4.7 1M High" }, + { id: "claude-opus-4-7-high-fast", name: "Opus 4.7 1M High Fast" }, + { id: "claude-opus-4-7-xhigh", name: "Opus 4.7 1M" }, + { id: "claude-opus-4-7-xhigh-fast", name: "Opus 4.7 1M Fast" }, + { id: "claude-opus-4-7-max", name: "Opus 4.7 1M Max" }, + { id: "claude-opus-4-7-max-fast", name: "Opus 4.7 1M Max Fast" }, + { id: "claude-opus-4-7-thinking-low", name: "Opus 4.7 1M Low Thinking" }, + { id: "claude-opus-4-7-thinking-low-fast", name: "Opus 4.7 1M Low Thinking Fast" }, + { id: "claude-opus-4-7-thinking-medium", name: "Opus 4.7 1M Medium Thinking" }, + { id: "claude-opus-4-7-thinking-medium-fast", name: "Opus 4.7 1M Medium Thinking Fast" }, + { id: "claude-opus-4-7-thinking-high", name: "Opus 4.7 1M High Thinking" }, + { id: "claude-opus-4-7-thinking-high-fast", name: "Opus 4.7 1M High Thinking Fast" }, + { id: "claude-opus-4-7-thinking-xhigh", name: "Opus 4.7 1M Thinking" }, + { id: "claude-opus-4-7-thinking-xhigh-fast", name: "Opus 4.7 1M Thinking Fast" }, + { id: "claude-opus-4-7-thinking-max", name: "Opus 4.7 1M Max Thinking" }, + { id: "claude-opus-4-7-thinking-max-fast", name: "Opus 4.7 1M Max Thinking Fast" }, + { id: "gpt-5.4-low", name: "GPT-5.4 1M Low" }, + { id: "gpt-5.4-medium", name: "GPT-5.4 1M" }, + { id: "gpt-5.4-medium-fast", name: "GPT-5.4 Fast" }, + { id: "gpt-5.4-high", name: "GPT-5.4 1M High" }, + { id: "gpt-5.4-high-fast", name: "GPT-5.4 High Fast" }, + { id: "gpt-5.4-xhigh", name: "GPT-5.4 1M Extra High" }, + { id: "gpt-5.4-xhigh-fast", name: "GPT-5.4 Extra High Fast" }, + { id: "claude-4.6-opus-high", name: "Opus 4.6 1M" }, + { id: "claude-4.6-opus-max", name: "Opus 4.6 1M Max" }, + { id: "claude-4.6-opus-high-thinking", name: "Opus 4.6 1M Thinking" }, + { id: "claude-4.6-opus-max-thinking", name: "Opus 4.6 1M Max Thinking" }, + { id: "claude-4.5-opus-high", name: "Opus 4.5" }, + { id: "claude-4.5-opus-high-thinking", name: "Opus 4.5 Thinking" }, + { id: "gpt-5.2-low", name: "GPT-5.2 Low" }, + { id: "gpt-5.2-low-fast", name: "GPT-5.2 Low Fast" }, + { id: "gpt-5.2-fast", name: "GPT-5.2 Fast" }, + { id: "gpt-5.2-high", name: "GPT-5.2 High" }, + { id: "gpt-5.2-high-fast", name: "GPT-5.2 High Fast" }, + { id: "gpt-5.2-xhigh", name: "GPT-5.2 Extra High" }, + { id: "gpt-5.2-xhigh-fast", name: "GPT-5.2 Extra High Fast" }, + { id: "gpt-5.6-luna-none", name: "GPT-5.6 Luna 1M None" }, + { id: "gpt-5.6-luna-none-fast", name: "GPT-5.6 Luna None Fast" }, + { id: "gpt-5.6-luna-low", name: "GPT-5.6 Luna 1M Low" }, + { id: "gpt-5.6-luna-low-fast", name: "GPT-5.6 Luna Low Fast" }, + { id: "gpt-5.6-luna-medium", name: "GPT-5.6 Luna 1M" }, + { id: "gpt-5.6-luna-medium-fast", name: "GPT-5.6 Luna Fast" }, + { id: "gpt-5.6-luna-high", name: "GPT-5.6 Luna 1M High" }, + { id: "gpt-5.6-luna-high-fast", name: "GPT-5.6 Luna High Fast" }, + { id: "gpt-5.6-luna-xhigh", name: "GPT-5.6 Luna 1M Extra High" }, + { id: "gpt-5.6-luna-xhigh-fast", name: "GPT-5.6 Luna Extra High Fast" }, + { id: "gpt-5.6-luna-max", name: "GPT-5.6 Luna 1M Max" }, + { id: "gpt-5.6-luna-max-fast", name: "GPT-5.6 Luna Max Fast" }, + { id: "gemini-3.6-flash-minimal", name: "Gemini 3.6 Flash Minimal" }, + { id: "gemini-3.6-flash-low", name: "Gemini 3.6 Flash Low" }, + { id: "gemini-3.6-flash-medium", name: "Gemini 3.6 Flash Medium" }, + { id: "gemini-3.6-flash-high", name: "Gemini 3.6 Flash" }, + { id: "gpt-5.4-mini-none", name: "GPT-5.4 Mini None" }, + { id: "gpt-5.4-mini-low", name: "GPT-5.4 Mini Low" }, + { id: "gpt-5.4-mini-medium", name: "GPT-5.4 Mini" }, + { id: "gpt-5.4-mini-high", name: "GPT-5.4 Mini High" }, + { id: "gpt-5.4-mini-xhigh", name: "GPT-5.4 Mini Extra High" }, + { id: "gpt-5.4-nano-none", name: "GPT-5.4 Nano None" }, + { id: "gpt-5.4-nano-low", name: "GPT-5.4 Nano Low" }, + { id: "gpt-5.4-nano-medium", name: "GPT-5.4 Nano" }, + { id: "gpt-5.4-nano-high", name: "GPT-5.4 Nano High" }, + { id: "gpt-5.4-nano-xhigh", name: "GPT-5.4 Nano Extra High" }, + { id: "claude-4.5-sonnet", name: "Sonnet 4.5" }, + { id: "claude-4.5-sonnet-thinking", name: "Sonnet 4.5 Thinking" }, + { id: "gpt-5.1-low", name: "GPT-5.1 Low" }, + { id: "gpt-5.1", name: "GPT-5.1" }, + { id: "gpt-5.1-high", name: "GPT-5.1 High" }, + { id: "gemini-3.5-flash", name: "Gemini 3.5 Flash" }, + { id: "claude-4-sonnet", name: "Sonnet 4" }, + { id: "claude-4-sonnet-thinking", name: "Sonnet 4 Thinking" }, + { id: "gpt-5-mini", name: "GPT-5 Mini" }, + { id: "kimi-k3-low", name: "Kimi K3 Low" }, + { id: "kimi-k3-max", name: "Kimi K3" }, + { id: "glm-5.2-high", name: "GLM 5.2" }, + { id: "glm-5.2-max", name: "GLM 5.2 Max" }, ], }; /** diff --git a/open-sse/config/providers/registry/dify/index.ts b/open-sse/config/providers/registry/dify/index.ts index de1d5b03cc5..80c9c7690fb 100644 --- a/open-sse/config/providers/registry/dify/index.ts +++ b/open-sse/config/providers/registry/dify/index.ts @@ -5,7 +5,11 @@ export const difyProvider: RegistryEntry = { alias: "dify", format: "openai", executor: "default", - baseUrl: "https://api.dify.ai/v1/chat/completions", + // Dify does not serve /chat/completions — its native completion route is + // POST /v1/chat-messages (validated via the dedicated dify validator, #11002). + // Keep this as the bare API root so route suffixes build correctly and + // self-hosted instances can override the base URL per connection. + baseUrl: "https://api.dify.ai", authType: "apikey", authHeader: "bearer", models: [{ id: "auto", name: "Auto" }], diff --git a/open-sse/config/providers/registry/hackclub/index.ts b/open-sse/config/providers/registry/hackclub/index.ts deleted file mode 100644 index 272ee5f86ca..00000000000 --- a/open-sse/config/providers/registry/hackclub/index.ts +++ /dev/null @@ -1,19 +0,0 @@ -import type { RegistryEntry } from "../../shared.ts"; - -export const hackclubProvider: RegistryEntry = { - id: "hackclub", - alias: "hc", - format: "openai", - executor: "default", - baseUrl: "https://ai.hackclub.com/proxy/v1/chat/completions", - modelsUrl: "https://ai.hackclub.com/proxy/v1/models", - authType: "optional", - authHeader: "bearer", - passthroughModels: true, - defaultContextLength: 128000, - models: [ - { id: "meta-llama/llama-3.3-70b-instruct", name: "Llama 3.3 70B" }, - { id: "mistralai/mistral-7b-instruct", name: "Mistral 7B" }, - { id: "deepseek-ai/deepseek-coder-33b", name: "DeepSeek Coder 33B" }, - ], -}; diff --git a/open-sse/config/providers/registry/kimi/web/index.ts b/open-sse/config/providers/registry/kimi/web/index.ts index 4344194974a..ddeea4350ba 100644 --- a/open-sse/config/providers/registry/kimi/web/index.ts +++ b/open-sse/config/providers/registry/kimi/web/index.ts @@ -12,10 +12,9 @@ export const kimi_webProvider: RegistryEntry = { alias: "kimi-web", format: "openai", executor: "kimi-web", - // International consumer chat — the legacy `kimi.moonshot.cn` domain now - // redirects every non-CN visitor to www.kimi.com, which speaks a different - // Connect-RPC API. See `open-sse/executors/kimi-web.ts` for the wire format. - baseUrl: "https://www.kimi.com", + // International consumer chat — Connect-RPC API at www.kimi.ai. + // See `open-sse/executors/kimi-web.ts` for the wire format. + baseUrl: "https://www.kimi.ai", authType: "apikey", authHeader: "Authorization", // Curated-only catalog. Agent Swarm is excluded because it requires Kimi's diff --git a/open-sse/config/providers/registry/minimax/web/index.ts b/open-sse/config/providers/registry/minimax/web/index.ts index 53ac1e6fe3f..6c2addc043b 100644 --- a/open-sse/config/providers/registry/minimax/web/index.ts +++ b/open-sse/config/providers/registry/minimax/web/index.ts @@ -16,7 +16,7 @@ export const hailuo_webProvider: RegistryEntry = { alias: "hailuo-web", format: "openai", executor: "hailuo-web", - baseUrl: "https://www.hailuo.ai", + baseUrl: "https://chat.minimax.io", authType: "apikey", authHeader: "bearer", models: HAILUO_WEB_STATIC_MODELS, diff --git a/open-sse/config/providers/registry/opencode/go/index.ts b/open-sse/config/providers/registry/opencode/go/index.ts index bf0c94ed90b..abebd92c0fc 100644 --- a/open-sse/config/providers/registry/opencode/go/index.ts +++ b/open-sse/config/providers/registry/opencode/go/index.ts @@ -1,4 +1,5 @@ import type { RegistryEntry } from "../../../shared.ts"; +import { OPENCODE_ZEN_GO_SHARED_MODELS } from "../../../shared.ts"; export const opencode_goProvider: RegistryEntry = { id: "opencode-go", @@ -23,9 +24,13 @@ export const opencode_goProvider: RegistryEntry = { { id: "glm-5.2", name: "GLM-5.2", supportsReasoning: true }, { id: "glm-5.2-high", name: "GLM-5.2 (high effort)", supportsReasoning: true }, { id: "glm-5.2-max", name: "GLM-5.2 (max effort)", supportsReasoning: true }, + + ...OPENCODE_ZEN_GO_SHARED_MODELS, + // models[0] (glm-5.2) is the dashboard default (LlmChatCard/ProviderTestSlideOver take models[0]). + { id: "glm-5.1", name: "GLM-5.1" }, { id: "glm-5", name: "GLM-5" }, - { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, + // kimi-k2.7-code declared identically on opencode-zen — see OPENCODE_ZEN_GO_SHARED_MODELS. { id: "kimi-k2.6", name: "Kimi K2.6" }, { id: "kimi-k2.5", name: "Kimi K2.5" }, // #8353: Kimi K3 base + max-effort alias from the OpenCode Go registry. @@ -89,7 +94,8 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: false, supportsReasoning: true, }, - { id: "qwen3.6-plus", name: "Qwen3.6 Plus", targetFormat: "claude", supportsVision: false }, + // qwen3.6-plus / qwen3.5-plus base ids declared identically on opencode-zen — see + // OPENCODE_ZEN_GO_SHARED_MODELS. { id: "qwen3.6-plus-high", name: "Qwen3.6 Plus (high effort)", @@ -104,7 +110,6 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: false, supportsReasoning: true, }, - { id: "qwen3.5-plus", name: "Qwen3.5 Plus", targetFormat: "claude", supportsVision: false }, // #8353: hy3 is the Go-tier base id (distinct from hy3-preview / hy3-free). { id: "hy3", name: "Hunyuan3", contextLength: 256000, supportsReasoning: true }, { @@ -138,6 +143,7 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: true, supportsAudio: true, supportsVideo: true, + targetFormat: "openai-responses", }, { id: "muse-spark-1.2-contributor-minimal", @@ -148,6 +154,7 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: true, supportsAudio: true, supportsVideo: true, + targetFormat: "openai-responses", }, { id: "muse-spark-1.2-contributor-low", @@ -158,6 +165,7 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: true, supportsAudio: true, supportsVideo: true, + targetFormat: "openai-responses", }, { id: "muse-spark-1.2-contributor-medium", @@ -168,6 +176,7 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: true, supportsAudio: true, supportsVideo: true, + targetFormat: "openai-responses", }, { id: "muse-spark-1.2-contributor-high", @@ -178,6 +187,7 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: true, supportsAudio: true, supportsVideo: true, + targetFormat: "openai-responses", }, { id: "muse-spark-1.2-contributor-xhigh", @@ -188,6 +198,7 @@ export const opencode_goProvider: RegistryEntry = { supportsVision: true, supportsAudio: true, supportsVideo: true, + targetFormat: "openai-responses", }, // #8353: Grok 4.5 + effort tiers from the OpenCode Go registry. { id: "grok-4.5", name: "Grok 4.5", supportsReasoning: true }, diff --git a/open-sse/config/providers/registry/opencode/zen/index.ts b/open-sse/config/providers/registry/opencode/zen/index.ts index 73da9c5dc51..9fdca2fc0a5 100644 --- a/open-sse/config/providers/registry/opencode/zen/index.ts +++ b/open-sse/config/providers/registry/opencode/zen/index.ts @@ -1,4 +1,5 @@ import type { RegistryEntry } from "../../../shared.ts"; +import { OPENCODE_ZEN_GO_SHARED_MODELS } from "../../../shared.ts"; export const opencode_zenProvider: RegistryEntry = { id: "opencode-zen", @@ -25,6 +26,10 @@ export const opencode_zenProvider: RegistryEntry = { supportsReasoning: true, interleavedField: "reasoning_content", }, + + ...OPENCODE_ZEN_GO_SHARED_MODELS, + // models[0] (big-pickle) is the dashboard default; SHARED spread kept after it. + { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, { id: "gpt-5.6-luna", name: "GPT 5.6 Luna" }, @@ -51,7 +56,27 @@ export const opencode_zenProvider: RegistryEntry = { { id: "grok-4.6", name: "Grok 4.6" }, // ── Muse ─────────────────────────────────────────────────── - { id: "muse-spark-1.2", name: "Muse Spark 1.2" }, + // Muse Spark is served by OpenCode Zen only on the OpenAI Responses API + // endpoint, not /chat/completions (see the opencode provider's own + // muse-spark entries, #10874/#10867) — this provider is a separate + // registry entry for the same upstream and never got the same + // targetFormat declaration, so requests routed here still hit + // /chat/completions with a mismatched or unanswerable body and the + // upstream returns an empty message. + { + id: "muse-spark-1.2", + name: "Muse Spark 1.2", + supportsReasoning: true, + targetFormat: "openai-responses", + }, + // Explicit wire-format overlay of the base opencode provider's muse-spark entry + // (targetFormat: openai-responses). Keep in sync with base on catalog syncs. + { + id: "muse-spark-1.2-contributor-free", + name: "Muse Spark 1.2 Contributor Free", + supportsReasoning: true, + targetFormat: "openai-responses", + }, // ── DeepSeek ──────────────────────────────────────────────── { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" }, @@ -66,7 +91,7 @@ export const opencode_zenProvider: RegistryEntry = { // ── Kimi / Moonshot ──────────────────────────────────────── { id: "kimi-k3", name: "Kimi K3" }, - { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, + // kimi-k2.7-code declared identically on opencode-go — see OPENCODE_ZEN_GO_SHARED_MODELS. // ── Qwen ─────────────────────────────────────────────────── // Issue #2292: Qwen models return Claude-format SSE bodies even @@ -74,8 +99,8 @@ export const opencode_zenProvider: RegistryEntry = { // through /messages and the Claude translator. // Issue #2822: These models are text-only — supportsVision: false // ensures combo routing skips them on image-bearing requests. - { id: "qwen3.5-plus", name: "Qwen3.5 Plus", targetFormat: "claude", supportsVision: false }, - { id: "qwen3.6-plus", name: "Qwen3.6 Plus", targetFormat: "claude", supportsVision: false }, + // qwen3.5-plus / qwen3.6-plus declared identically on opencode-go — see + // OPENCODE_ZEN_GO_SHARED_MODELS. // ── Free Tier ────────────────────────────────────────────── // #6998 (2026-07-14): upstream free tier rotated — minimax-m2.5-free, diff --git a/open-sse/config/providers/registry/zcode/index.ts b/open-sse/config/providers/registry/zcode/index.ts index cd2a4eece64..65e2e1c3d3d 100644 --- a/open-sse/config/providers/registry/zcode/index.ts +++ b/open-sse/config/providers/registry/zcode/index.ts @@ -1,6 +1,17 @@ import type { RegistryEntry } from "../../shared.ts"; import { GLM_SHARED_MODELS } from "../../../glmProvider.ts"; +const GLM_EXECUTOR_EFFORT_ALIASES = new Set([ + "glm-5.3-high", + "glm-5.3-low", + "glm-5.2-high", + "glm-5.2-max", +]); + +export const ZCODE_MODELS = GLM_SHARED_MODELS.filter( + (model) => !GLM_EXECUTOR_EFFORT_ALIASES.has(model.id) +).map((model) => ({ ...model, supportedThinkingEfforts: [] })); + /** * Local ZCode app-server backend. Authentication remains in the user's local * ZCode profile (`builtin:zai-coding-plan`); OmniRoute does not receive or @@ -14,5 +25,7 @@ export const zcodeProvider: RegistryEntry = { baseUrl: "zcode://app-server/stdio", authType: "none", authHeader: "none", - models: [...GLM_SHARED_MODELS], + // ZCode's app-server transport does not consume reasoning_effort; keep thinking + // capability metadata without advertising aliases or tiers that it would ignore. + models: ZCODE_MODELS, }; diff --git a/open-sse/config/providers/shared.ts b/open-sse/config/providers/shared.ts index d87250644dc..62909a528d5 100644 --- a/open-sse/config/providers/shared.ts +++ b/open-sse/config/providers/shared.ts @@ -25,6 +25,7 @@ import { GLMT_TIMEOUT_MS, GLM_SHARED_MODELS, } from "../glmProvider.ts"; +import { OPENCODE_ZEN_GO_SHARED_MODELS } from "../opencodeZenGoSharedModels.ts"; import { MARITALK_DEFAULT_BASE_URL } from "../maritalk.ts"; import { CURSOR_REGISTRY_VERSION, @@ -719,6 +720,7 @@ export { GLM_TIMEOUT_MS, GLMT_TIMEOUT_MS, GLM_SHARED_MODELS, + OPENCODE_ZEN_GO_SHARED_MODELS, MARITALK_DEFAULT_BASE_URL, CURSOR_REGISTRY_VERSION, getAntigravityProviderHeaders, diff --git a/open-sse/config/searchRegistry.ts b/open-sse/config/searchRegistry.ts index 9230fa1b0e3..28136655fe2 100644 --- a/open-sse/config/searchRegistry.ts +++ b/open-sse/config/searchRegistry.ts @@ -10,6 +10,8 @@ * perplexity-search reuses credentials from the "perplexity" chat provider. */ +import { isProviderBlockedByIdOrAlias } from "@/shared/utils/noAuthProviders"; + export interface SearchProviderConfig { id: string; name: string; @@ -280,19 +282,44 @@ export const SEARCH_PROVIDERS: Record = { cacheTTLMs: 5 * 60 * 1000, fallbackOnly: true, }, + + // SuperGrok / xAI server-side X Search. Not web search. Explicit provider or + // search_type "x" only — never auto-selected for generic web queries. + "x-search": { + id: "x-search", + name: "X Search (Grok)", + baseUrl: "https://api.x.ai/v1/responses", + method: "POST", + authType: "apikey", + authHeader: "bearer", + costPerQuery: 0, + freeMonthlyQuota: 0, + searchTypes: ["x"], + defaultMaxResults: 5, + maxMaxResults: 20, + timeoutMs: 60_000, + cacheTTLMs: 5 * 60 * 1000, + }, }; /** * Credential fallback mapping — search providers that can reuse credentials * from a related provider (e.g., perplexity-search uses the same API key as perplexity chat). */ -export const SEARCH_CREDENTIAL_FALLBACKS: Record = { +export const SEARCH_CREDENTIAL_FALLBACKS: Record = { "perplexity-search": "perplexity", "ollama-search": "ollama-cloud", "zai-search": "zai", "jina-search": "jina-ai", + "x-search": ["xai-oauth", "xao", "xai"], }; +export function getSearchCredentialFallbacks(providerId: string): string[] { + const mapped = SEARCH_CREDENTIAL_FALLBACKS[providerId]; + if (!mapped) return []; + return Array.isArray(mapped) ? mapped : [mapped]; +} + /** * Request-only aliases for POST /v1/search. * @@ -316,6 +343,8 @@ export const SEARCH_PROVIDER_ALIASES: Record = { searxng: "searxng-search", zai: "zai-search", duckduckgo: "duckduckgo-free", + "x_search": "x-search", + x: "x-search", }; export function resolveSearchProviderId(providerId: string): string { @@ -327,6 +356,24 @@ export function resolveSearchProviderId(providerId: string): string { * Request routing should use resolveSearchProvider() so aliases work * without colliding with the Foundation jina-ai provider id. */ +const CATALOG_SEARXNG_DEFAULT_URL = "http://localhost:8888/search"; + +/** + * Catalog default SearXNG URL is a desktop convenience. In Docker/K8s nothing + * listens on :8888, and OMNIROUTE_ALLOW_PRIVATE_PROVIDER_URLS (needed for + * ClusterIP providers) lets ProxyFetch attempt it, producing ECONNREFUSED and + * a 502 that then burns the next fallback's quota. Skip unless the operator + * overrode baseUrl. + */ +export function isUnconfiguredLoopbackSearchProvider( + provider: SearchProviderConfig | null | undefined +): boolean { + if (!provider || provider.id !== "searxng-search") return false; + const configured = String(provider.baseUrl || "").replace(/\/+$/, ""); + const catalog = CATALOG_SEARXNG_DEFAULT_URL.replace(/\/+$/, ""); + return configured === catalog; +} + export function getSearchProvider(providerId: string): SearchProviderConfig | null { return SEARCH_PROVIDERS[providerId] || null; } @@ -349,16 +396,18 @@ export function supportsSearchType( /** * Get all search providers as a flat list */ -export function getAllSearchProviders(): Array<{ +export function getAllSearchProviders(blockedProviders: string[] = []): Array<{ id: string; name: string; searchTypes: string[]; }> { - return Object.values(SEARCH_PROVIDERS).map((p) => ({ - id: p.id, - name: p.name, - searchTypes: p.searchTypes, - })); + return Object.values(SEARCH_PROVIDERS) + .filter((p) => !p.disabled && !isProviderBlockedByIdOrAlias(p.id, blockedProviders)) + .map((p) => ({ + id: p.id, + name: p.name, + searchTypes: p.searchTypes, + })); } /** @@ -379,10 +428,11 @@ export function selectProvider( // Auto-selection excludes fallbackOnly providers so a free cost-0 provider never // overrides a configured paid one — they are reached only via explicit id or the - // route handler's last-resort step. + // route handler's last-resort step. Missing searchType follows the API default + // (`web`) so X-only providers are never cheapest-wins for generic queries. + const effectiveType = searchType || "web"; const providers = Object.values(SEARCH_PROVIDERS).filter( - (provider) => - !provider.fallbackOnly && (searchType ? supportsSearchType(provider, searchType) : true) + (provider) => !provider.fallbackOnly && supportsSearchType(provider, effectiveType) ); if (providers.length === 0) return null; diff --git a/open-sse/executors/accountRotation.ts b/open-sse/executors/accountRotation.ts index b5f25afc66d..67115bdd8d0 100644 --- a/open-sse/executors/accountRotation.ts +++ b/open-sse/executors/accountRotation.ts @@ -1,7 +1,7 @@ /** * Shared multi-account rotation mechanics for noauth executors that round-robin * across several "accounts" (fingerprints), each with an optional dedicated - * proxy — currently `OpencodeExecutor` and `MimocodeExecutor`. + * proxy — currently `OpencodeExecutor`. * * Extracted after both executors independently implemented the same * pickAccount/markCooldown/markSuccess skeleton with the same exponential @@ -40,6 +40,15 @@ export interface RotatableAccount { cooldownUntil: number; consecutiveFails: number; proxy: AccountProxyConfig["proxy"]; + evictedAt?: number | null; +} + +export type CooldownKind = "transient" | "terminal"; + +const EVICT_AFTER_TERMINAL = 3; + +export function isAccountEvicted(account: RotatableAccount): boolean { + return account.evictedAt != null; } const COOLDOWN_BASE_MS = TRANSIENT_COOLDOWN_MS; @@ -74,17 +83,21 @@ export function pickAccount( return accounts[fallbackIdx]; } -export function markCooldown(account: RotatableAccount): void { +export function markCooldown(account: RotatableAccount, kind: CooldownKind = "transient"): void { account.consecutiveFails++; const backoff = Math.min( COOLDOWN_BASE_MS * Math.pow(2, account.consecutiveFails - 1), COOLDOWN_MAX_MS ); account.cooldownUntil = Date.now() + backoff + Math.random() * 1000; + if (kind === "terminal" && account.consecutiveFails >= EVICT_AFTER_TERMINAL) { + account.evictedAt = Date.now(); + } } export function markSuccess(account: RotatableAccount): void { account.consecutiveFails = 0; + account.evictedAt = null; } /** Mask an account id for logs (UI calls it a fingerprint). */ @@ -107,3 +120,58 @@ export function maskAccountId(fingerprint: string): string { export function isNetworkErrorRotatable(account: RotatableAccount): boolean { return account.proxy !== null; } + +/** + * Detect an *empty* upstream rejection: a 400 whose body carries no usable + * completion — the kind `OpencodeExecutor` must rotate/retry on instead of + * propagating as a fatal success. + * + * Signature is deliberately strict and scoped to the observed malformed + * envelope (`choices[0].message` with no `error`, no real `content`, + * `finish_reason: null`): + * - status must be exactly 400 (anything else → false); + * - body must parse and contain a `choices` array with at least one entry + * holding a `message` object; + * - an `error` field (present or empty) → false, so genuine 400s keep + * propagating immediately (#10460 precedent: classify by signature before + * rotating); + * - `tool_calls` / `reasoning_content` → false (real content); + * - `message.content` absent / null / "" → eligible; any other value + * (non-empty text, number, block array…) → false (conservative); + * - a literal `finish_reason` (not null) → false (a completed, if empty, turn). + * + * Does NOT reuse `detectMalformedNonStream` (diagnostics.ts): that classifier + * also flags `{error:{…}}` bodies as `empty_choices`, which would rotate on + * real errors — a false-positive class with a history here. + */ +export function isEmptyUpstreamRejection(status: number, bodyText: string): boolean { + if (status !== 400) return false; + let parsed: unknown; + try { + parsed = JSON.parse(bodyText); + } catch { + return false; + } + const choices = (parsed as { choices?: unknown })?.choices; + if (!Array.isArray(choices) || choices.length === 0) return false; + const first = choices[0] as { message?: unknown; finish_reason?: unknown }; + if (typeof first !== "object" || first === null) return false; + const rawMessage = (first as { message?: unknown }).message; + if (typeof rawMessage === "undefined" || rawMessage === null) return false; + if (typeof parsed !== "object" || parsed === null) return false; + if ("error" in (parsed as Record)) return false; + const msg = rawMessage as Record; + if ("tool_calls" in msg) return false; + if ("reasoning_content" in msg) return false; + const content = msg.content; + if (content !== undefined && content !== null && content !== "") return false; + if (first.finish_reason !== null && first.finish_reason !== undefined) return false; + return true; +} + +/** Best-effort extraction of the upstream `chatcmpl_*` id from a response body, + * for observability logging. Returns `"unknown"` when absent or unparseable. */ +export function extractChatcmplId(bodyText: string): string { + const match = /"id"\s*:\s*"(chatcmpl_[^"]+)"/.exec(bodyText); + return match ? match[1] : "unknown"; +} diff --git a/open-sse/executors/base.ts b/open-sse/executors/base.ts index 877431b68e5..1c13442aff0 100644 --- a/open-sse/executors/base.ts +++ b/open-sse/executors/base.ts @@ -20,6 +20,10 @@ import { recordLearnedThinkingCap, parseThinkingBudgetMax, } from "../services/learnedThinkingCaps.ts"; +import { + recordLearnedReasoningEffort, + parseReasoningEffortEnum, +} from "../services/learnedReasoningEffortCaps.ts"; import { getParamFilterConfig, addParamToBlocklist, @@ -104,6 +108,12 @@ import { import { applyPeerTraceHeader } from "@/shared/resilience/peerRouting"; import { applyClineProtocolHeaders } from "@/shared/utils/clineAuth"; import { isProbeContext } from "@/shared/utils/probeOrigin"; +import { + parseAndValidatePublicUrl, + parseAndValidateNonMetadataUrl, +} from "@/shared/network/outboundUrlGuard"; +import { getProviderValidationGuard } from "@/shared/network/outboundUrlGuardPolicy"; +import { isLocalProvider, isSelfHostedChatProvider } from "@/shared/constants/providers"; // Header helpers extracted to a pure leaf; re-exported for external importers // (executors + tests) that import them from "./base.ts". export { @@ -397,6 +407,29 @@ export class BaseExecutor { return fallback || this.config.baseUrl || ""; } + /** + * SSRF guard for the runtime dispatch path (GHSA-4f49-hj64-448x). A persisted, + * caller-supplied `providerSpecificData.baseUrl` reaches the fetch() calls + * below, so a `manage`-scope actor (or, on a keyless install, an anonymous + * one) could point a provider at loopback / internal / cloud-metadata hosts + * and exfiltrate the stored upstream key. Mirror the provider VALIDATION + * guard so runtime dispatch makes the same decision the validation layer + * already makes: local / self-hosted providers are exempt (they legitimately + * use private URLs, and the OMNIROUTE_ALLOW_PRIVATE_PROVIDER_URLS opt-in still + * applies through the guard), and for everything else `public-only` mode + * blocks private + metadata while the default `block-metadata` mode blocks the + * cloud-metadata IMDS pivot. Throws on a blocked URL. + */ + protected assertOutboundUrlAllowed(url: string): void { + if (!url) return; + if (isLocalProvider(this.provider) || isSelfHostedChatProvider(this.provider)) return; + if (getProviderValidationGuard() === "public-only") { + parseAndValidatePublicUrl(url); + return; + } + parseAndValidateNonMetadataUrl(url); + } + /** * Alternate protocol selected on this connection, if the provider declares one * that matches. Centralizes the registry lookup so every call-site resolves the @@ -615,6 +648,7 @@ export class BaseExecutor { async countTokens({ model, body, credentials, signal, log }: CountTokensInput) { const url = this.buildCountTokensUrl(model, credentials); if (!url) return null; + this.assertOutboundUrlAllowed(url); // GHSA-4f49 const headers = this.buildHeaders(credentials, false); const requestBody = @@ -796,6 +830,9 @@ export class BaseExecutor { // loop. The learned cap is also recorded process-wide via // recordLearnedThinkingCap so future requests skip the 400 entirely. let thinkingBudgetClampedMax: number | null = null; + // Set by the reasoning_effort 4xx clamp-and-retry below — guards the same + // "fires at most once per URL" invariant as thinkingBudgetClampedMax above. + let reasoningEffortClamped = false; for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) { const requestCredentials = withForcedResponsesUpstream( @@ -869,6 +906,9 @@ export class BaseExecutor { // Timeout only covers response start; stream stalls are handled downstream. const fetchStartTimeoutMs = this.getTimeoutMs(); const fetchWithStartTimeout = async (requestUrl: string, requestOptions: RequestInit) => { + // GHSA-4f49: guard here (not only next to the first buildUrl) so retries + // and fallback URLs are validated too, before any bytes leave the host. + this.assertOutboundUrlAllowed(requestUrl); const timeoutController = fetchStartTimeoutMs > 0 ? new AbortController() : null; let timeoutId: ReturnType | null = null; if (timeoutController) { @@ -1496,6 +1536,49 @@ export class BaseExecutor { } } + // Reasoning-effort enum 4xx clamp-and-retry (any provider/model without a + // declared reasoning_effort capability — custom OpenAI-compatible + // connections, or a registered provider the registry hasn't caught up + // with). Mirrors the thinking_budget clamp-and-retry above: parse the + // upstream-advertised accepted values, record them process-wide (so + // FUTURE requests clamp proactively via sanitizeReasoningEffortForProvider + // → getLearnedReasoningEffort), clamp the live transformedBody by + // re-running the sanitizer, and retry the same URL once. + if ( + (response.status === HTTP_STATUS.BAD_REQUEST || + response.status === HTTP_STATUS.UNPROCESSABLE_ENTITY) && + !reasoningEffortClamped && + transformedBody && + typeof transformedBody === "object" + ) { + const errText = await response + .clone() + .text() + .catch(() => ""); + const acceptedValues = parseReasoningEffortEnum(errText); + if (acceptedValues) { + reasoningEffortClamped = true; + const learned = recordLearnedReasoningEffort(this.provider, model, acceptedValues); + if (learned) { + transformedBody = sanitizeReasoningEffortForProvider( + transformedBody, + this.provider, + model, + log + ); + let retryBody = JSON.stringify(transformedBody); + if (usesClaudeCodeProtocol || this.provider === "claude") { + retryBody = await signRequestBody(retryBody); + } + log?.info?.( + "REASONING_SANITIZE", + `Upstream ${response.status} rejected reasoning_effort on ${url} — clamped to ${learned} and retrying (learned for ${this.provider}/${model})` + ); + response = await fetchWithStartTimeout(url, { ...fetchOptions, body: retryBody }); + } + } + } + // Generic reactive 400 field-downgrade; each field is stripped at most once. if ( response.status === HTTP_STATUS.BAD_REQUEST && diff --git a/open-sse/executors/base/reasoningEffort.ts b/open-sse/executors/base/reasoningEffort.ts index fa416dbc1a8..8dd99904fd6 100644 --- a/open-sse/executors/base/reasoningEffort.ts +++ b/open-sse/executors/base/reasoningEffort.ts @@ -8,6 +8,10 @@ import { getProviderModel, getProviderModels, } from "../../config/providerModels.ts"; +import { + getLearnedReasoningEffort, + REASONING_EFFORT_ORDER, +} from "../../services/learnedReasoningEffortCaps.ts"; /** * Sanitize reasoning_effort for providers that don't accept all values. @@ -338,10 +342,24 @@ export function sanitizeReasoningEffortForProvider( const supportsXHigh = supportsXHighEffort(provider, modelStr); const supportsMax = supportsMaxEffortForProvider(provider, modelStr); + // Highest value we've actually seen this provider+model accept in a real + // upstream 4xx (learnedReasoningEffortCaps.ts) — takes priority over the + // static registry (which defaults to "supports everything" when there's no + // entry, e.g. custom OpenAI-compatible connections) and over the hardcoded + // "high" fallback below (which isn't always valid either). + const learnedCap = getLearnedReasoningEffort(provider, modelStr); + const learnedRank = learnedCap ? REASONING_EFFORT_ORDER.indexOf(learnedCap) : -1; // ── xhigh handling ────────────────────────────────────────────────────── // xhigh is OmniRoute-internal. Map it to the best effort the model accepts. if (effortStr === "xhigh") { + if (learnedCap && learnedRank < REASONING_EFFORT_ORDER.indexOf("xhigh")) { + log?.info?.( + "REASONING_SANITIZE", + `${provider}/${modelStr}: clamped reasoning_effort xhigh → ${learnedCap} (learned)` + ); + return writeEffortValue(b, learnedCap, c); + } if (supportsXHigh) return body; // model accepts xhigh natively if (supportsMax) { log?.info?.( @@ -366,6 +384,13 @@ export function sanitizeReasoningEffortForProvider( // upstream, and if it 400s the user gets a clear signal. This prevents // new models from being unusable for weeks until they're whitelisted (#8057). if (effortStr === "max") { + if (learnedCap && learnedRank < REASONING_EFFORT_ORDER.indexOf("max")) { + log?.info?.( + "REASONING_SANITIZE", + `${provider}/${modelStr}: clamped reasoning_effort max → ${learnedCap} (learned)` + ); + return writeEffortValue(b, learnedCap, c); + } if (supportsMax) return body; // explicitly known to accept max // A model that explicitly advertises its accepted tiers is safe to normalize. diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index ab595c3cbb6..6e39bb81caa 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -481,15 +481,18 @@ function toCodexResponseFailedEvent(parsed: Record): Record({ transform(chunk, controller) { buffer += decoder.decode(chunk, { stream: true }); - let sep: number; - while ((sep = buffer.indexOf("\n\n")) !== -1) { - const block = buffer.slice(0, sep + 2); - buffer = buffer.slice(sep + 2); + while (true) { + const separator = /\r?\n\r?\n/.exec(buffer); + if (!separator) break; + const blockEnd = separator.index + separator[0].length; + const block = buffer.slice(0, blockEnd); + buffer = buffer.slice(blockEnd); if (!dropBlock(block)) controller.enqueue(encoder.encode(block)); } }, @@ -701,8 +706,8 @@ export function encodeResponseSseEvent(raw: string): { sse: string; terminal: bo // "Invalid state: Controller is already closed". The earlier empty-payload // check below never caught codex.rate_limits — over WS the frame carries a // non-empty JSON payload (`{"type":"codex.rate_limits", ...}`), so - // `!payload.trim()` is false. Match by event type instead. Opt-in via - // OMNIROUTE_CODEX_DROP_NONSTANDARD_EVENTS (the HTTP transport is handled + // `!payload.trim()` is false. Match by event type instead. Default ON via + // OMNIROUTE_CODEX_DROP_NONSTANDARD_EVENTS (#11014); the HTTP transport is handled // separately by filterNonstandardCodexSse, since super.execute forwards the // upstream stream verbatim and never runs this function). if (eventType.startsWith("codex.") && codexDropNonstandardEvents()) { diff --git a/open-sse/executors/commandCode.ts b/open-sse/executors/commandCode.ts index b736e6cdf1f..a1023fc80cd 100644 --- a/open-sse/executors/commandCode.ts +++ b/open-sse/executors/commandCode.ts @@ -1,6 +1,3 @@ -import { randomUUID } from "node:crypto"; - -import { isVisionModelId } from "@/shared/constants/visionModels"; import { REGISTRY } from "../config/providerRegistry.ts"; import { BaseExecutor, @@ -11,407 +8,54 @@ import { type JsonRecord = Record; -export const COMMAND_CODE_VERSION = process.env.COMMAND_CODE_VERSION?.trim() || "1.15.1"; -// Hard server-side ceiling enforced by Command Code's /alpha/generate endpoint: -// any request with params.max_tokens > 200_000 is rejected with a 400 -// "Too big: expected number to be <=200000 at params.max_tokens". We only use -// this to clamp a CLIENT-SUPPLIED max_tokens down to a value the endpoint will -// accept; we never fabricate this number for requests that omit the field (see -// clampMaxTokens / buildCommandCodeBody). +// Defensive server-side ceiling for a CLIENT-SUPPLIED max_tokens. The official +// /provider/v1/chat/completions endpoint (documented OpenAI-format surface) is +// the successor to the CLI-only /alpha/generate endpoint, which rejected any +// params.max_tokens > 200_000 with a 400. We only clamp a client-supplied value +// down; we never fabricate this number for requests that omit the field (see +// clampMaxTokens / buildOpenAiBody). const MAX_COMMAND_CODE_TOKENS = 200_000; -const encoder = new TextEncoder(); function isRecord(value: unknown): value is JsonRecord { return typeof value === "object" && value !== null && !Array.isArray(value); } -function asRecordArray(value: unknown): JsonRecord[] { - return Array.isArray(value) ? value.filter(isRecord) : []; -} - -function stringValue(value: unknown): string | undefined { - return typeof value === "string" ? value : undefined; -} - function numberValue(value: unknown): number | undefined { return typeof value === "number" && Number.isFinite(value) ? value : undefined; } -function recordOrEmpty(value: unknown): JsonRecord { - if (isRecord(value)) return value; - if (typeof value === "string" && value.trim()) { - try { - const parsed: unknown = JSON.parse(value); - if (isRecord(parsed)) return parsed; - } catch (error) { - console.warn( - "[commandCode] tool arg parse failed:", - error instanceof Error ? error.message : String(error) - ); - } - } - return {}; -} - -/** - * Build the `arguments` field for an assistant tool-call part that Command - * Code's /alpha/generate schema REQUIRES (rejects a missing field with - * `missing required field 'arguments'`). Valid source values round-trip: - * - object arguments -> JSON string of the object - * - valid JSON string arguments -> the string as-is - * - missing / empty / invalid JSON string -> "{}" (a valid empty-object string) - */ -function toolCallArgumentsString(value: unknown): string { - if (isRecord(value)) return JSON.stringify(value); - if (typeof value === "string" && value.trim()) { - try { - const parsed: unknown = JSON.parse(value); - if (isRecord(parsed)) return value; - } catch { - return "{}"; - } - return "{}"; - } - return JSON.stringify(recordOrEmpty(value)); -} - -/** - * Tool names that collide with Command Code's server-side built-in tools. - * The /alpha/generate server normalizes tool-call/tool-result parts against - * ITS OWN built-in registry for matching names; for its built-in `tool_search` - * the result normalization requires `arguments` in a shape we do not send, so - * the result is rejected with `input[N] missing required field 'arguments'` - * (verified live 2026-08-10 — renaming the call/result `tool_search` → `grep` - * makes the identical request pass; the server pairs each tool-result with the - * nearest preceding tool-call, so any result following such a call is affected). - * We rename the colliding name consistently on the wire — definitions, calls - * and results — then un-rename on the response path so the client still sees - * its original tool names. - */ -const COMMAND_CODE_RESERVED_TOOL_NAMES = new Set(["tool_search"]); - -function wireToolName(clientName: string, toolNameMap: Map): string { - if (COMMAND_CODE_RESERVED_TOOL_NAMES.has(clientName)) { - const wire = `omniroute_${clientName}`; - toolNameMap.set(wire, clientName); - return wire; - } - return clientName; -} - -function clientToolName(wireName: string, toolNameMap: Map): string { - return toolNameMap.get(wireName) ?? wireName; -} - -function normalizeContentText(content: unknown): string { - if (typeof content === "string") return content; - return asRecordArray(content) - .filter((part) => part.type === "text") - .map((part) => stringValue(part.text) || "") - .join("\n"); -} - -/** - * Model id patterns for Command Code models that have `text, vision` - * capability per the official CC model registry, but are NOT caught - * by the shared {@link isVisionModelId} heuristic. Kept as a local - * set because these are CC-specific model IDs (vendor-prefix shapes - * like "moonshotai/Kimi-K2.6" or CC aliases like "gpt-5.6-luna"). - * - * Source: Command Code /alpha/generate model registry (docs). - */ -const CC_VISION_MODEL_PATTERNS: readonly RegExp[] = [ - // Open Source - /kimi-k2/i, // moonshotai/Kimi-K2.6, Kimi-K2.7-Code, Kimi-K2.5 - /qwen3\.\d/i, // Qwen/Qwen3.6-Plus, Qwen/Qwen3.7-Plus - /step-?3/i, // stepfun/Step-3.7-Flash - // Anthropic - /claude-fable/i, // claude-fable-5 (not covered by claude-opus/sonnet/haiku-4) - // OpenAI - /gpt-5/i, // gpt-5.6, gpt-5.5, gpt-5.4, gpt-5.4-mini, gpt-5.3-codex - // NOTE: gpt-5.4-mini and gpt-5.3-codex deliberately stay inside the `/gpt-5/` - // family — both accept image input on the OpenAI API, and there is no - // verified Command Code backend data marking them text-only. Excluding them - // without evidence would re-create #4071 (image stripped from a model that - // can see it). Revisit only with per-model CC registry capability data. - // Sakana - /fugu/i, // sakana/fugu-ultra -]; - -/** - * Whether a model id routed through the Command Code executor is - * vision-capable. Checks Mimo-specific rules first, then CC-specific - * patterns, then falls through to the shared {@link isVisionModelId} - * heuristic (which covers minimax-m3, claude-3/4 families, gemini, - * gpt-4o/4.1, mistral-medium-3, and general "-vision" / "multimodal"). - */ -function isCommandCodeVisionModel(model?: string | null): boolean { - if (!model) return false; - // mimo-v2.5-pro is text-only — exclude before any positive check - if (/(?:^|\/)mimo-v2\.5-pro$/i.test(model)) return false; - // Only mimo-v2.5 and mimo-v2-omni accept images per Xiaomi vendor docs - if (/(?:^|\/)mimo-v2\.5$/i.test(model)) return true; - if (/(?:^|\/)mimo-v2-omni$/i.test(model)) return true; - // CC-specific patterns: Kimi K2, Qwen 3.x, Stepfun, Claude Fable, - // GPT-5, Sakana Fugu — not covered by the shared heuristic - if (CC_VISION_MODEL_PATTERNS.some((pattern) => pattern.test(model))) return true; - // Fall through: minimax-m3, claude-3/4, gemini-2/3, gpt-4o, -vision, multimodal - return isVisionModelId(model); -} - -/** - * Extract the image URL from an OpenAI-compatible or Command Code - * content part, returning undefined for non-image parts. - * - * OpenAI-compatible: { type: "image_url", image_url: { url: "..." } } - * Command Code CLI: { type: "image", image: "..." } - * AI SDK image: { type: "image", image: "data:...;base64,..." } (#1330) - * Anthropic image: { type: "image", source: { type: "base64", media_type, data } } - * or { type: "image", source: { type: "url", url } } - * - * The Anthropic-shaped block is common for Claude-Code-compatible clients - * (e.g. Zoo Code) that send Messages-style content arrays to the - * OpenAI `/v1/chat/completions` surface. Without this branch the image was - * silently dropped before reaching the upstream vision model. - */ -function extractImageUrl(part: JsonRecord): string | undefined { - if (part.type === "image") { - const direct = stringValue(part.image); - if (direct) return direct; - - // Anthropic source block: { source: { type: "base64", media_type, data } } or - // { source: { type: "url", url } }. - const source = isRecord(part.source) ? part.source : null; - if (source) { - if (source.type === "base64") { - const mediaType = stringValue(source.media_type) || "image/png"; - const data = stringValue(source.data); - if (data) return `data:${mediaType};base64,${data}`; - } - if (source.type === "url") { - const url = stringValue(source.url); - if (url) return url; - } - } - return undefined; - } - if (part.type === "image_url") { - if (isRecord(part.image_url)) return stringValue(part.image_url.url); - return stringValue(part.image_url); - } - return undefined; -} - -/** - * Convert an OpenAI-format content array to Command Code's internal - * CLI format. For vision-capable models (MiniMax M3, MiMo v2.5, etc.) - * this also preserves image parts alongside text. - */ -function convertUserContentParts(content: unknown, isVisionModel: boolean): string | unknown[] { - // For non-vision models or string content, extract text only. - if (!isVisionModel || typeof content === "string") { - return normalizeContentText(content); - } - - const parts: unknown[] = []; - for (const part of asRecordArray(content)) { - if (part.type === "text") { - const text = stringValue(part.text); - if (text) parts.push({ type: "text", text }); - continue; - } - const imgUrl = extractImageUrl(part); - if (imgUrl) { - parts.push({ type: "image", image: imgUrl }); - continue; - } - // Always drop tool_use / tool_result / thinking parts from user - // messages (Command Code doesn't accept them for role:"user"). - } - - // When every part was stripped, fall back to empty text so the - // message is still valid JSON (Command Code rejects empty content). - if (parts.length === 0) parts.push({ type: "text", text: "" }); - - return parts; -} - -function convertTools(tools: unknown, toolNameMap: Map): unknown[] { - return asRecordArray(tools).map((tool) => { - const fn = isRecord(tool.function) ? tool.function : tool; - return { - type: "function", - name: wireToolName(stringValue(fn.name) || "", toolNameMap), - description: stringValue(fn.description) || "", - input_schema: isRecord(fn.parameters) ? fn.parameters : {}, - }; - }); -} - -function buildToolCallMetadata( - messages: JsonRecord[], - toolNameMap: Map -): { - pairedToolCallIds: Set; - toolCallNames: Map; - toolCallArgs: Map; -} { - const callIds = new Set(); - const resultIds = new Set(); - const toolCallNames = new Map(); - const toolCallArgs = new Map(); - - for (const message of messages) { - if (message.role === "assistant") { - for (const call of asRecordArray(message.tool_calls)) { - const id = stringValue(call.id); - if (id) { - callIds.add(id); - const fn = isRecord(call.function) ? call.function : {}; - const name = stringValue(fn.name) || stringValue(call.name); - if (name) toolCallNames.set(id, wireToolName(name, toolNameMap)); - toolCallArgs.set(id, toolCallArgumentsString(fn.arguments)); - } - } - } else if (message.role === "tool") { - const id = stringValue(message.tool_call_id); - if (id) resultIds.add(id); - } - } - - const pairedToolCallIds = new Set([...callIds].filter((id) => resultIds.has(id))); - return { pairedToolCallIds, toolCallNames, toolCallArgs }; -} - -function convertMessages( - messages: unknown, - model?: string | null, - toolNameMap?: Map -): { system: string; messages: unknown[] } { - const source = asRecordArray(messages); - const { pairedToolCallIds, toolCallNames, toolCallArgs } = buildToolCallMetadata( - source, - toolNameMap ?? new Map() - ); - const out: unknown[] = []; - const system: string[] = []; - const isVision = isCommandCodeVisionModel(model); - - for (const message of source) { - const role = stringValue(message.role); - if (role === "system" || role === "developer") { - const text = normalizeContentText(message.content); - if (text) system.push(text); - continue; - } - - if (role === "user") { - out.push({ role: "user", content: convertUserContentParts(message.content, isVision) }); - continue; - } - - if (role === "assistant") { - const parts: unknown[] = []; - const text = normalizeContentText(message.content); - if (text) parts.push({ type: "text", text }); - - for (const call of asRecordArray(message.tool_calls)) { - const id = stringValue(call.id) || ""; - if (!id || !pairedToolCallIds.has(id)) continue; - const fn = isRecord(call.function) ? call.function : {}; - const parsedInput = recordOrEmpty(fn.arguments); - parts.push({ - type: "tool-call", - toolCallId: id, - toolName: wireToolName( - stringValue(fn.name) || stringValue(call.name) || "unknown", - toolNameMap ?? new Map() - ), - input: parsedInput, - // /alpha/generate requires this field on assistant tool-call parts; - // a missing one is rejected with `missing required field 'arguments'`. - arguments: toolCallArgumentsString(fn.arguments), - }); - } - - if (parts.length > 0) out.push({ role: "assistant", content: parts }); - continue; - } - - if (role === "tool") { - const toolCallId = stringValue(message.tool_call_id) || ""; - if (!toolCallId || !pairedToolCallIds.has(toolCallId)) continue; - const toolName = wireToolName( - stringValue(message.name) || toolCallNames.get(toolCallId) || "unknown", - toolNameMap ?? new Map() - ); - out.push({ - role: "tool", - content: [ - { - type: "tool-result", - toolCallId, - toolName, - // /alpha/generate requires `arguments` here too (same rejection as - // tool-call parts); echo the paired call's args, defensively "{}". - arguments: toolCallArgs.get(toolCallId) ?? "{}", - output: { type: "text", value: normalizeContentText(message.content) }, - }, - ], - }); - } - } - - return { system: system.join("\n\n"), messages: out }; -} - // Clamp a client-supplied max_tokens to the endpoint ceiling, mirroring the // provider-driven clamp in antigravity.ts: we only intervene when the value is -// present, positive AND would otherwise be rejected (> 200_000). A valid value -// is returned floored; anything absent, non-numeric or non-positive returns -// undefined so the caller can OMIT the field entirely and let Command Code's -// upstream apply the model's own native default (rather than us inventing a -// number). A non-positive value such as Zoo Code's max_tokens:-1 ("let the -// server choose") must be omitted, NOT forced to 1 — the old Math.max(1,...) -// truncated output to a single token (#5166). +// present, positive AND would otherwise be rejected (> MAX_COMMAND_CODE_TOKENS). +// A valid value is returned floored; anything absent, non-numeric or non-positive +// returns undefined so the caller can OMIT the field entirely and let the +// provider's upstream apply the model's own native default (rather than us +// inventing a number). A non-positive value such as Zoo Code's max_tokens:-1 +// ("let the server choose") must be omitted, NOT forced to 1 — the old +// Math.max(1,...) truncated output to a single token (#5166). function clampMaxTokens(value: unknown): number | undefined { const numeric = numberValue(value); if (numeric === undefined || numeric <= 0) return undefined; return Math.min(Math.floor(numeric), MAX_COMMAND_CODE_TOKENS); } -// Reasoning/thinking fields that payload rules or clients may inject and that -// CommandCode's upstream accepts inside `params`. Without this pass-through, -// payload-rule overrides on these fields are silently dropped (#2986 follow-up). -const COMMAND_CODE_PASSTHROUGH_FIELDS = [ - "reasoning_effort", - "reasoning", - "thinking", - "effort", - "output_config", - "extra_body", -] as const; - /** - * Command Code's /alpha/generate endpoint serves most models under a - * vendor-prefixed wire id (e.g. `xiaomi/mimo-v2.5`, `deepseek/deepseek-v4-pro`, - * `moonshotai/Kimi-K2.6`) and defaults an unprefixed id to the `anthropic:` - * provider, which 403s with "Model/provider not recognized: anthropic:". + * Command Code serves most models under a vendor-prefixed wire id (e.g. + * `xiaomi/mimo-v2.5`, `deepseek/deepseek-v4-pro`, `moonshotai/Kimi-K2.6`). * The command-code registry ids already carry the vendor prefix, so a bare id * reaching the executor is an operator-set custom model (e.g. the Vision Bridge * picker, #10809). Map the small set of documented bare ids to their * vendor-prefixed wire form; anything with an explicit `/` (or already wired) - * passes through untouched. Kept minimal and doc-backed, mirroring the - * `CC_VISION_MODEL_PATTERNS` philosophy. + * passes through untouched. Kept minimal and doc-backed. */ const COMMAND_CODE_BARE_MODEL_VENDOR_PREFIX: Readonly> = { - // Xiaomi MiMo V2.5 — the only CC-served vision model not in the registry. + // Xiaomi MiMo V2.5 — a CC-served vision model not in the registry. "mimo-v2.5": "xiaomi/mimo-v2.5", "mimo-v2.5-pro": "xiaomi/mimo-v2.5-pro", }; /** - * Normalize an incoming model id to the wire form Command Code's upstream + * Normalize an incoming model id to the wire form Command Code's provider API * accepts. Strips a leading provider prefix (`command-code/` / `cmd/`) that the * pipeline may have resolved, then maps known bare ids to their * vendor-prefixed form (see above). @@ -424,546 +68,47 @@ function normalizeCommandCodeWireModel(model: string): string { return COMMAND_CODE_BARE_MODEL_VENDOR_PREFIX[bare] ?? bare; } -function buildCommandCodeBody( +/** + * Build a flat OpenAI chat.completions request body for the official + * /provider/v1/chat/completions endpoint. The incoming body is already the + * standard OpenAI chat.completions shape (registry `format: "openai"`), so this + * is a passthrough that: normalizes the wire model id, forces the stream flag + * to match the caller's expectation, clamps max_tokens, and lets reasoning / + * payload-rule passthrough fields flow through untouched. No CLI envelope + * (config/memory/taste/skills/permissionMode) and no CLI-shaped message + * conversion here — /provider/v1 is the documented, standard API. + */ +function buildOpenAiBody( model: string, body: unknown, - stream = false -): { body: JsonRecord; toolNameMap: Map } { - const input = isRecord(body) ? body : {}; - const toolNameMap = new Map(); + stream: boolean +): { body: JsonRecord } { + const input = isRecord(body) ? { ...(body as JsonRecord) } : {}; - // Payload rules may rewrite `body.model` (e.g. deepseek-v4-pro-max → - // deepseek/deepseek-v4-pro for the command-code provider). Prefer the - // rewritten value if present; fall back to the resolved combo model arg. - // Normalize to the vendor-prefixed wire id the upstream requires (#10809). const resolvedModel = normalizeCommandCodeWireModel( - typeof input.model === "string" && input.model.trim().length > 0 ? input.model : model + typeof input.model === "string" && input.model.trim().length > 0 + ? input.model + : model ); - const converted = convertMessages(input.messages, resolvedModel, toolNameMap); - const explicitSystem = typeof input.system === "string" ? input.system : ""; - const system = [converted.system, explicitSystem].filter(Boolean).join("\n\n"); - - const params: JsonRecord = { + const out: JsonRecord = { + ...input, model: resolvedModel, - messages: converted.messages, - tools: convertTools(input.tools, toolNameMap), - system, - stream: true, + stream: stream === true, }; - // Only forward max_tokens when the client actually supplied one. Omitting it - // lets Command Code's upstream apply the model's own native default, so we - // never invent a value (the old behavior, which sent the wrong number and got - // DeepSeek V4 rejected with "Too big: expected number to be <=200000"). When - // present, it is clamped to the endpoint ceiling so an oversized client value - // degrades gracefully instead of 400ing. + // Forward max_tokens only when the client actually supplied a positive value + // (clamped to the endpoint ceiling). Omitting it lets the provider's upstream + // apply the model's own native default; a non-positive value such as -1 + // ("let the server choose") must be omitted, NOT coerced to 1 (#5166). const maxTokens = clampMaxTokens(input.max_tokens ?? input.max_completion_tokens); + delete out.max_tokens; + delete out.max_completion_tokens; if (maxTokens !== undefined) { - params.max_tokens = maxTokens; - } - - for (const field of COMMAND_CODE_PASSTHROUGH_FIELDS) { - const value = input[field]; - if (value !== undefined && value !== null) { - params[field] = value; - } - } - - return { - body: { - config: { - workingDir: "/workspace", - date: new Date().toISOString().slice(0, 10), - environment: "external", - structure: [], - isGitRepo: false, - currentBranch: "", - mainBranch: "", - gitStatus: "", - recentCommits: [], - }, - memory: "", - taste: "", - skills: "", - permissionMode: "standard", - params, - }, - toolNameMap, - }; -} - -function parseStreamLine(line: string): unknown | undefined { - let trimmed = line.trim(); - if (!trimmed || trimmed.startsWith(":") || trimmed.startsWith("event:")) return undefined; - if (trimmed.startsWith("data:")) trimmed = trimmed.slice(5).trim(); - if (!trimmed || trimmed === "[DONE]") return undefined; - - try { - return JSON.parse(trimmed); - } catch (error) { - console.warn( - "[commandCode] stream line parse failed:", - error instanceof Error ? error.message : String(error) - ); - return undefined; - } -} - -function mapFinishReason(reason: unknown): "stop" | "length" | "tool_calls" { - if (reason === "tool-calls" || reason === "tool_calls" || reason === "toolUse") - return "tool_calls"; - if ( - reason === "length" || - reason === "max_tokens" || - reason === "max-tokens" || - reason === "max_output_tokens" - ) { - return "length"; - } - return "stop"; -} - -function chatCompletionChunk( - id: string, - model: string, - delta: JsonRecord, - finishReason: unknown = null -) { - return { - id, - object: "chat.completion.chunk", - created: Math.floor(Date.now() / 1000), - model, - choices: [{ index: 0, delta, finish_reason: finishReason }], - }; -} - -function sse(data: unknown): Uint8Array { - return encoder.encode(`data: ${JSON.stringify(data)}\n\n`); -} - -type AggregateState = { - content: string; - reasoning: string; - toolCalls: JsonRecord[]; - finishReason: "stop" | "length" | "tool_calls"; - usage: JsonRecord | null; -}; - -function firstRecord(record: JsonRecord, keys: readonly string[]): JsonRecord { - for (const key of keys) { - const value = record[key]; - if (isRecord(value)) return value; - } - return {}; -} - -function firstNumber(record: JsonRecord, keys: readonly string[]): number | undefined { - for (const key of keys) { - const value = numberValue(record[key]); - if (value !== undefined) return value; - } - return undefined; -} - -/** Keep earlier finish-step usage when the terminal finish event omits it. */ -function mergeCommandCodeUsage(previous: JsonRecord | null, next: unknown): JsonRecord | null { - if (!isRecord(next)) return previous; - - const merged: JsonRecord = { ...(previous || {}), ...next }; - for (const key of [ - "inputTokenDetails", - "input_token_details", - "input_tokens_details", - "prompt_tokens_details", - "outputTokenDetails", - "output_token_details", - "output_tokens_details", - "completion_tokens_details", - "reasoningTokenDetails", - "reasoning_token_details", - ]) { - const before = isRecord(previous?.[key]) ? previous[key] : {}; - const after = isRecord(next[key]) ? next[key] : {}; - if (Object.keys(before).length > 0 || Object.keys(after).length > 0) { - merged[key] = { ...before, ...after }; - } - } - return merged; -} - -function rememberCommandCodeUsage(state: AggregateState, event: JsonRecord): void { - const usage = - event.type === "finish-step" - ? (event.usage ?? event.totalUsage) - : (event.totalUsage ?? event.usage); - state.usage = mergeCommandCodeUsage(state.usage, usage); -} - -function applyEventToAggregate( - event: JsonRecord, - state: AggregateState, - toolNameMap: Map -): void { - // Some Command Code protocol revisions attach usage to the terminal payload - // without preserving the event type. Capture it before event-specific handling. - rememberCommandCodeUsage(state, event); - - switch (event.type) { - case "text-delta": - state.content += stringValue(event.text) || ""; - break; - case "reasoning-delta": - state.reasoning += stringValue(event.text) || ""; - break; - case "tool-call": { - const args = recordOrEmpty(event.input ?? event.args ?? event.arguments); - state.toolCalls.push({ - id: stringValue(event.toolCallId) || stringValue(event.id) || randomUUID(), - type: "function", - function: { - name: clientToolName( - stringValue(event.toolName) || stringValue(event.name) || "", - toolNameMap - ), - arguments: JSON.stringify(args), - }, - }); - break; - } - case "finish-step": - break; - case "finish": - state.finishReason = mapFinishReason(event.finishReason); - break; - } -} - -function applyEventToAggregateOrThrow( - event: JsonRecord, - state: AggregateState, - toolNameMap: Map -): void { - if (event.type === "error") { - const error = isRecord(event.error) ? event.error : {}; - throw new Error( - stringValue(error.message) || stringValue(event.error) || "Command Code stream error" - ); + out.max_tokens = maxTokens; } - applyEventToAggregate(event, state, toolNameMap); -} - -function usageFromCommandCode(usage: JsonRecord | null) { - if (!usage) return undefined; - const inputDetails = firstRecord(usage, [ - "inputTokenDetails", - "input_token_details", - "input_tokens_details", - "prompt_tokens_details", - ]); - const outputDetails = firstRecord(usage, [ - "outputTokenDetails", - "output_token_details", - "output_tokens_details", - "completion_tokens_details", - ]); - const reasoningDetails = firstRecord(usage, [ - "reasoningTokenDetails", - "reasoning_token_details", - "reasoning_tokens_details", - ]); - const cacheRead = - firstNumber(usage, [ - "cachedInputTokens", - "cached_input_tokens", - "cacheReadInputTokens", - "cache_read_input_tokens", - "cacheReadTokens", - "cache_read_tokens", - "cached_tokens", - ]) ?? - firstNumber(inputDetails, [ - "cachedTokens", - "cached_tokens", - "cacheReadTokens", - "cache_read_tokens", - ]); - const noCache = firstNumber(inputDetails, ["noCacheTokens", "no_cache_tokens"]); - // Command Code's totalUsage.inputTokens is the FULL prompt total and already - // includes the cached portion (noCacheTokens + cacheReadTokens = inputTokens), - // so we must NOT add cacheRead back — that would double-count. There is no - // cache-write field in the upstream payload, so cache creation stays unset. - const prompt = - firstNumber(usage, ["inputTokens", "input_tokens", "promptTokens", "prompt_tokens"]) ?? - (noCache ?? 0) + (cacheRead ?? 0); - const reasoning = - firstNumber(usage, ["reasoningTokens", "reasoning_tokens"]) ?? - firstNumber(outputDetails, ["reasoningTokens", "reasoning_tokens"]) ?? - firstNumber(reasoningDetails, ["reasoningTokens", "reasoning_tokens"]); - const textOutput = firstNumber(outputDetails, ["textTokens", "text_tokens"]); - const completion = - firstNumber(usage, [ - "outputTokens", - "output_tokens", - "completionTokens", - "completion_tokens", - ]) ?? (textOutput ?? 0) + (reasoning ?? 0); - const total = firstNumber(usage, ["totalTokens", "total_tokens"]) ?? prompt + completion; - const result: JsonRecord = { - prompt_tokens: prompt, - prompt_tokens_details: { cached_tokens: cacheRead ?? 0 }, - completion_tokens: completion, - completion_tokens_details: { reasoning_tokens: reasoning ?? 0 }, - total_tokens: total, - }; - // Surface the cache breakdown as informational fields so logUsage prints - // `| cache_read=X | no_cache=Y` and appendRequestLog persists them. These are - // NOT added to prompt_tokens (already included) — metering stays accurate. - if (cacheRead !== undefined && cacheRead > 0) result.cache_read_input_tokens = cacheRead; - if (noCache !== undefined && noCache > 0) result.no_cache_tokens = noCache; - if (reasoning !== undefined && reasoning > 0) result.reasoning_tokens = reasoning; - return result; -} - -function createStreamResponse( - upstream: Response, - model: string, - signal?: AbortSignal | null, - toolNameMap: Map = new Map() -): Response { - const id = `chatcmpl-${randomUUID()}`; - const reader = upstream.body?.getReader(); - const decoder = new TextDecoder(); - let buffer = ""; - let sentRole = false; - let closed = false; - const state: AggregateState = { - content: "", - reasoning: "", - toolCalls: [], - finishReason: "stop", - usage: null, - }; - - const stream = new ReadableStream({ - start(controller) { - if (!reader) { - controller.error(new Error("Command Code response missing body")); - return; - } - - const abort = () => { - closed = true; - reader.cancel().catch(() => undefined); - controller.error(new DOMException("The operation was aborted", "AbortError")); - }; - signal?.addEventListener("abort", abort, { once: true }); - - const emitEvent = (event: unknown) => { - if (!isRecord(event) || closed) return; - rememberCommandCodeUsage(state, event); - if (!sentRole) { - sentRole = true; - controller.enqueue(sse(chatCompletionChunk(id, model, { role: "assistant" }))); - } - - switch (event.type) { - case "text-delta": { - const text = stringValue(event.text) || ""; - if (text) controller.enqueue(sse(chatCompletionChunk(id, model, { content: text }))); - state.content += text; - break; - } - case "reasoning-delta": { - const text = stringValue(event.text) || ""; - if (text) { - controller.enqueue(sse(chatCompletionChunk(id, model, { reasoning_content: text }))); - state.reasoning += text; - } - break; - } - case "tool-call": { - const index = state.toolCalls.length; - const args = recordOrEmpty(event.input ?? event.args ?? event.arguments); - const toolCall = { - id: stringValue(event.toolCallId) || stringValue(event.id) || randomUUID(), - type: "function", - function: { - name: clientToolName( - stringValue(event.toolName) || stringValue(event.name) || "", - toolNameMap - ), - arguments: JSON.stringify(args), - }, - }; - state.toolCalls.push(toolCall); - controller.enqueue( - sse(chatCompletionChunk(id, model, { tool_calls: [{ index, ...toolCall }] })) - ); - break; - } - case "reasoning-end": - break; - case "finish-step": - break; - case "finish": { - state.finishReason = mapFinishReason(event.finishReason); - controller.enqueue(sse(chatCompletionChunk(id, model, {}, state.finishReason))); - // Emit a standards-compliant usage-only chunk (choices: []) before - // [DONE] when upstream reported usage. stream.ts's extractUsage - // recognizes this shape (see stream.ts:1661) and logs the ACTUAL - // token counts (in/out/cache_read/no_cache) instead of estimates. - const usagePayload = usageFromCommandCode(state.usage); - if (usagePayload) { - controller.enqueue( - sse({ - id, - object: "chat.completion.chunk", - model, - usage: usagePayload, - choices: [], - }) - ); - } - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - closed = true; - controller.close(); - reader.cancel().catch(() => undefined); - break; - } - case "error": { - const error = isRecord(event.error) ? event.error : {}; - throw new Error( - stringValue(error.message) || stringValue(event.error) || "Command Code stream error" - ); - } - } - }; - - const pump = async () => { - try { - for (;;) { - if (closed) return; - const { done, value } = await reader.read(); - if (done) break; - buffer += decoder.decode(value, { stream: true }); - const lines = buffer.split("\n"); - buffer = lines.pop() || ""; - for (const line of lines) emitEvent(parseStreamLine(line)); - } - if (buffer.trim()) emitEvent(parseStreamLine(buffer)); - if (!closed) { - if (!sentRole) - controller.enqueue(sse(chatCompletionChunk(id, model, { role: "assistant" }))); - controller.enqueue(sse(chatCompletionChunk(id, model, {}, state.finishReason))); - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - controller.close(); - } - } catch (error) { - controller.error(error); - } finally { - signal?.removeEventListener("abort", abort); - try { - reader.releaseLock(); - } catch (error) { - console.warn( - "[commandCode] reader releaseLock failed:", - error instanceof Error ? error.message : String(error) - ); - } - } - }; - - pump(); - }, - cancel() { - closed = true; - return reader?.cancel(); - }, - }); - - return new Response(stream, { - status: 200, - headers: { "Content-Type": "text/event-stream; charset=utf-8", "Cache-Control": "no-cache" }, - }); -} - -async function createJsonResponse( - upstream: Response, - model: string, - signal?: AbortSignal | null, - toolNameMap: Map = new Map() -): Promise { - const reader = upstream.body?.getReader(); - if (!reader) throw new Error("Command Code response missing body"); - - const decoder = new TextDecoder(); - let buffer = ""; - const state: AggregateState = { - content: "", - reasoning: "", - toolCalls: [], - finishReason: "stop", - usage: null, - }; - - try { - for (;;) { - if (signal?.aborted) throw new DOMException("The operation was aborted", "AbortError"); - const { done, value } = await reader.read(); - if (done) break; - buffer += decoder.decode(value, { stream: true }); - const lines = buffer.split("\n"); - buffer = lines.pop() || ""; - for (const line of lines) { - const event = parseStreamLine(line); - if (!isRecord(event)) continue; - applyEventToAggregateOrThrow(event, state, toolNameMap); - } - } - if (buffer.trim()) { - const event = parseStreamLine(buffer); - if (isRecord(event)) applyEventToAggregateOrThrow(event, state, toolNameMap); - } - } finally { - try { - await reader.cancel(); - } catch (error) { - console.warn( - "[commandCode] reader cancel failed:", - error instanceof Error ? error.message : String(error) - ); - } - try { - reader.releaseLock(); - } catch (error) { - console.warn( - "[commandCode] reader releaseLock failed:", - error instanceof Error ? error.message : String(error) - ); - } - } - - const message: JsonRecord = { role: "assistant", content: state.content }; - if (state.reasoning) message.reasoning_content = state.reasoning; - if (state.toolCalls.length > 0) message.tool_calls = state.toolCalls; - - const payload: JsonRecord = { - id: `chatcmpl-${randomUUID()}`, - object: "chat.completion", - created: Math.floor(Date.now() / 1000), - model, - choices: [{ index: 0, message, finish_reason: state.finishReason }], - }; - const usage = usageFromCommandCode(state.usage); - if (usage) payload.usage = usage; - - return new Response(JSON.stringify(payload), { - status: 200, - headers: { "Content-Type": "application/json" }, - }); + return { body: out }; } export class CommandCodeExecutor extends BaseExecutor { @@ -973,7 +118,7 @@ export class CommandCodeExecutor extends BaseExecutor { buildUrl() { const baseUrl = (this.config.baseUrl || "https://api.commandcode.ai").replace(/\/$/, ""); - return `${baseUrl}${this.config.chatPath || "/alpha/generate"}`; + return `${baseUrl}${this.config.chatPath || "/provider/v1/chat/completions"}`; } async execute({ model, body, stream, credentials, signal, upstreamExtraHeaders }: ExecuteInput) { @@ -983,26 +128,17 @@ export class CommandCodeExecutor extends BaseExecutor { const headers: Record = { "Content-Type": "application/json", Authorization: `Bearer ${apiKey}`, - "x-command-code-version": COMMAND_CODE_VERSION, - "x-cli-environment": "external", - "x-project-slug": "pi-cc", - "x-taste-learning": "false", - "x-co-flag": "false", - "x-session-id": randomUUID(), + Accept: stream ? "text/event-stream" : "application/json", }; mergeUpstreamExtraHeaders(headers, upstreamExtraHeaders); // The combo/single-model dispatch boundary does not always run // sanitizeRequestForResolvedTarget before reaching this executor (combo // path), and Command Code rejects unsupported reasoning_effort values - // outright (e.g. "minimal" → 400 "expected one of low|medium|high|xhigh|max"). - // Sanitize here — the executor is the last line of defense for the wire body. + // outright. Sanitize here — the executor is the last line of defense for + // the wire body. const sanitizedBody = sanitizeReasoningEffortForProvider(body, this.provider, model); - const { body: transformedBody, toolNameMap } = buildCommandCodeBody( - model, - sanitizedBody, - stream - ); + const { body: transformedBody } = buildOpenAiBody(model, sanitizedBody, stream); const url = this.buildUrl(); const upstream = await fetch(url, { method: "POST", @@ -1028,10 +164,9 @@ export class CommandCodeExecutor extends BaseExecutor { }; } - const response = stream - ? createStreamResponse(upstream, model, signal, toolNameMap) - : await createJsonResponse(upstream, model, signal, toolNameMap); - - return { response, url, headers, transformedBody }; + // The /provider/v1/chat/completions endpoint returns standard OpenAI-format + // SSE (stream) or JSON (non-stream) straight through, so the upstream + // Response passes through untouched — no AI-SDK/CLI event re-parsing needed. + return { response: upstream, url, headers, transformedBody }; } -} +} \ No newline at end of file diff --git a/open-sse/executors/copilot-m365-connection.ts b/open-sse/executors/copilot-m365-connection.ts index c8251af5f4c..1463a4da632 100644 --- a/open-sse/executors/copilot-m365-connection.ts +++ b/open-sse/executors/copilot-m365-connection.ts @@ -165,8 +165,7 @@ export function resolveConnectionParams( // token is an opaque JWE with 5) is the freshest copy: the executor refreshes it // in place before resolving params, and the framework mutates it after a refresh. const credentialsJwt = - typeof credentials?.accessToken === "string" && - credentials.accessToken.split(".").length === 3 + typeof credentials?.accessToken === "string" && credentials.accessToken.split(".").length === 3 ? credentials.accessToken : ""; const accessToken = @@ -309,9 +308,7 @@ export function tokenNeedsRefresh(token: string, leadMs = M365_REFRESH_LEAD_MS): } /** The freshest readable access token for a connection (JWT column → apiKey → psd). */ -export function currentM365AccessToken( - credentials: ProviderCredentials | undefined -): string { +export function currentM365AccessToken(credentials: ProviderCredentials | undefined): string { if ( typeof credentials?.accessToken === "string" && credentials.accessToken.split(".").length === 3 @@ -393,21 +390,169 @@ export async function refreshM365AccessToken( } } -/** Flatten OpenAI messages into a single prompt (system instructions prepended). */ -export function buildPrompt(body: JsonRecord | undefined): string { - const messages = (body?.messages as Array) || []; - const systemMsgs = messages.filter((m) => m.role === "system"); - const userMsg = messages.filter((m) => m.role === "user").pop(); - const userText = - typeof userMsg?.content === "string" ? userMsg.content : JSON.stringify(userMsg?.content ?? ""); - let prompt = ""; - if (systemMsgs.length > 0) { - const sysText = systemMsgs - .map((m) => (typeof m.content === "string" ? m.content : "")) +/** A client-declared tool, normalized from the OpenAI `tools[]` entry. */ +export interface M365ToolSpec { + name: string; + description: string; + parameters: JsonRecord | null; +} + +/** + * Extract `tools` / `tool_choice` from an OpenAI chat-completion body, normalizing + * function tools into {@link M365ToolSpec}. Non-function tools and entries without + * a name are dropped (they cannot be expressed in the M365 protocol). + */ +export function extractToolSpec(body: JsonRecord | undefined): { + tools: M365ToolSpec[]; + toolChoice: unknown; +} { + const raw = Array.isArray(body?.tools) ? (body!.tools as JsonRecord[]) : []; + const tools: M365ToolSpec[] = []; + for (const t of raw) { + if (t?.type !== "function") continue; + const fn = (t.function ?? {}) as JsonRecord; + const name = typeof fn.name === "string" ? fn.name : ""; + if (!name) continue; + tools.push({ + name, + description: typeof fn.description === "string" ? fn.description : "", + parameters: + fn.parameters && typeof fn.parameters === "object" ? (fn.parameters as JsonRecord) : null, + }); + } + return { tools, toolChoice: body?.tool_choice ?? null }; +} + +/** Compact a tool result before it is folded into the flattened prompt. */ +function compactToolResult(text: string, maxChars = 4000): string { + if (text.length <= maxChars) return text; + return `${text.slice(0, maxChars)}\n…[truncated ${text.length - maxChars} chars]`; +} + +function messageText(content: unknown): string { + if (typeof content === "string") return content; + if (Array.isArray(content)) { + // Multimodal content parts: keep text parts, skip image parts (unsupported here). + return content + .map((p) => + p && typeof p === "object" && typeof (p as JsonRecord).text === "string" + ? (p as JsonRecord).text + : "" + ) .filter(Boolean) .join("\n"); - if (sysText) prompt += `[System Instructions]\n${sysText}\n\n`; } - prompt += userText; - return prompt; + return content == null ? "" : JSON.stringify(content); +} + +/** + * Flatten the FULL OpenAI message history into a single bracketed prompt — earlier + * turns, assistant replies (including `tool_calls`), and tool results, so multi-turn + * agent loops keep their context. Tool results are compacted via + * {@link compactToolResult} to keep a long loop from exhausting the turn budget. + */ +export function flattenMessages(body: JsonRecord | undefined): string { + const messages = (body?.messages as Array) || []; + const parts: string[] = []; + for (const m of messages) { + const role = typeof m.role === "string" ? m.role.toLowerCase().trim() : "user"; + const text = messageText(m.content).trim(); + if (Array.isArray(m.tool_calls) && m.tool_calls.length > 0) { + if (text) parts.push(`[${role}]\n${text}`); + parts.push(`[${role} tool_calls]\n${JSON.stringify(m.tool_calls)}`); + continue; + } + if (role === "tool") { + const id = typeof m.tool_call_id === "string" ? m.tool_call_id : ""; + parts.push(`[tool result id=${id}]\n${compactToolResult(text)}`); + continue; + } + if (!text) continue; + parts.push(`[${role}]\n${text}`); + } + return parts.join("\n\n").trim(); +} + +/** + * Wrap the flattened prompt in the community M365 tool-calling protocol: definitions + * inside a `` block, and the model answering with fenced blocks whose info + * string is the exact tool name and whose body is a JSON object of arguments. + * `tool_choice: "none"` keeps the plain prompt (no tool use requested this turn). + */ +function toolProtocolPrompt(text: string, tools: M365ToolSpec[], toolChoice: unknown): string { + if (tools.length === 0 || toolChoice === "none") { + return `Please answer the following request in full. Do not truncate or abbreviate your response.\n\n${text}`; + } + const defs = tools.map((t) => { + const params = t.parameters ? JSON.stringify(t.parameters, null, 2) : "{}"; + return `${t.name} — ${t.description}\n\`\`\`${t.name}\n${params}\n\`\`\``; + }); + return ( + `You are an execution agent operating on behalf of the application that sent this ` + + `request. The tools below are real, active, and callable right now — they were ` + + `registered by that application for this conversation. Do not analyze whether tools ` + + `are registered, available, or permitted: they are. Never state that a tool is ` + + `unavailable or that you cannot call tools.\n` + + `When the user's request requires a tool, call it by emitting one or more fenced code ` + + `blocks. Each block's info string is the exact tool name and its body is a single JSON ` + + `object of arguments. For independent operations, emit multiple blocks in one response. ` + + `Do not wrap tool calls in any other structure, and wait for the tool result before ` + + `claiming completion.\n\n\n${defs.join("\n\n")}\n\n\n${text}` + ); +} + +/** + * Flatten OpenAI messages into a single prompt (full history), and — when the + * client declared `tools` — wrap it in the M365 fenced-block tool protocol so the + * model's tool calls can be parsed back into OpenAI `tool_calls` downstream. + */ +export function buildPrompt(body: JsonRecord | undefined): string { + const { tools, toolChoice } = extractToolSpec(body); + return toolProtocolPrompt(flattenMessages(body), tools, toolChoice); +} + +/** + * Build the ROUTER-planning prompt — the strategy the substrate model actually + * complies with. Asking it to "use" a client tool gets refused ("not available in + * this chat environment") because it checks its own plugin registry; asking it to + * act as a tool-SELECTION assistant that prints a routing decision as plain text + * (`CALL_TOOL: name({...})` / `NO_TOOL_NEEDED`) bypasses that refusal entirely. + */ +export function buildRouterPrompt( + text: string, + tools: M365ToolSpec[], + toolChoice: unknown +): string { + const defs = JSON.stringify( + tools.map((t) => ({ + type: "function", + function: { name: t.name, description: t.description, parameters: t.parameters ?? {} }, + })) + ); + const choice = + typeof toolChoice === "string" && toolChoice !== "auto" && toolChoice !== "none" + ? toolChoice + : toolChoice && typeof toolChoice === "object" + ? (((toolChoice as JsonRecord).function as JsonRecord | undefined)?.name ?? "auto") + : "auto"; + let rules = + `- If a tool is needed, respond with: CALL_TOOL: tool_name({"arg1":"value1"})\n` + + `- If multiple independent tools are needed, output one CALL_TOOL line per tool\n` + + `- If no tool is needed, respond with: NO_TOOL_NEEDED\n` + + `- Only use tools from the available list above\n` + + `- Validate all arguments against the tool's schema\n` + + `- Do not invent tools that are not in the list`; + // Multi-turn: completed tool evidence in the history was already acted upon — + // re-invoking those tools would duplicate work. + if (text.includes("[tool result id=") || text.includes("[assistant tool_calls]")) { + rules += + `\n- Completed evidence must not be repeated: prior tool_calls/tool results are ` + + `already delivered, never re-invoke them\n` + + `- Only start a new tool call when fresh unfinished work remains on the current request`; + } + return ( + `You are a tool selection assistant. Based on the user request, decide which tool to call next.\n\n` + + `Available tools: ${defs}\n\nMODE: ${choice}\n\nRules:\n${rules}\n\n` + + `User request and evidence:\n${text}` + ); } diff --git a/open-sse/executors/copilot-m365-frames.ts b/open-sse/executors/copilot-m365-frames.ts index c8c6dee7be0..f4172b11904 100644 --- a/open-sse/executors/copilot-m365-frames.ts +++ b/open-sse/executors/copilot-m365-frames.ts @@ -18,6 +18,8 @@ * accumulated — NOT incremental) → isLastUpdate:true → type:2 final → type:3 completion. */ +type JsonRecord = Record; + /** SignalR record separator (0x1e) terminating every JSON frame. */ export const RECORD_SEPARATOR = String.fromCharCode(0x1e); @@ -28,17 +30,45 @@ export const HANDSHAKE_REQUEST = { protocol: "json", version: 1 } as const; export const KEEPALIVE_PING = { type: 6 } as const; /** - * Allowed message types observed in the 2026-08 recapture of the working - * `m365.cloud.microsoft/chat` client (#10718). The old 11-entry list is no longer - * seen on the wire — the stale shape gets closed immediately after the type:4. + * Allowed message types observed in a 2026-08-21 live capture of a working + * `m365.cloud.microsoft/chat` session (issue: "Stream ended before producing a + * non-ping SSE event" on every individual/consumer M365 Copilot call). The + * #10718 6-entry shape above no longer produces a `type:1 target:"update"` + * frame at all — the socket only replies with SignalR keepalive pings and then + * closes, which is exactly what surfaces client-side as that generic stream + * error. 30 entries, up from 6. */ export const ALLOWED_MESSAGE_TYPES = [ "Chat", "Suggestion", + "InternalSearchQuery", "Disengaged", + "InternalLoaderMessage", "Progress", + "GeneratedCode", + "RenderCardRequest", + "AdsQuery", + "SemanticSerp", + "GenerateContentQuery", + "GenerateGraphicArt", + "SearchQuery", + "ConfirmationCard", + "AuthError", + "DeveloperLogs", + "TriggerPlugin", + "HintInvocation", + "MemoryUpdate", "EndOfRequest", - "InternalLoaderMessage", + "TriggerConfirmation", + "ResumeInvokeAction", + "ResumeUserInputRequest", + "TriggerUserInputRequest", + "EscapeHatch", + "TriggerPluginAuth", + "ResumePluginAuth", + "SideBySide", + "ReferencesListComplete", + "SwitchRespondingEndpoint", ] as const; /** @@ -76,19 +106,26 @@ export const M365_ENTERPRISE_EXTRA_MESSAGE_TYPES = [ ] as const; /** - * Individual / EDU option sets from the 2026-08 recapture (#10718) — 14 entries. - * The previous 25-entry consumer/MSA set (enable_msa_user, pdnascan, cwc_code_*, - * …) is no longer observed on the wire and belongs to the shape the substrate - * now drops silently. + * Individual / EDU option sets from a 2026-08-21 live capture — 34 entries, up + * from the #10718 14-entry shape (which itself superseded an earlier 25-entry + * shape). Each recapture so far has been additive/reshuffled rather than a + * wholesale replacement — treat this as the protocol continuing to drift, not + * a one-time fix; a future capture may again need to update this list. */ export const M365_DEFAULT_OPTION_SETS = [ "search_result_progress_messages_with_search_queries", "update_textdoc_response_after_streaming", "deepleo_networking_timeout_10minutes_canmore", "cwc_flux_image", + "cwc_code_interpreter", + "cwc_code_interpreter_amsfix", "cwcfluxgptv", "flux_v3_gptv_enable_upload_multi_image_in_turn_wo_ch", "gptvnorm2048", + "cwc_code_interpreter_citation_fix", + "code_interpreter_interactive_charts", + "cwc_code_interpreter_interactive_charts_inline_image", + "code_interpreter_matplotlib_patching", "cwc_fileupload_odb", "update_memory_plugin", "add_custom_instructions", @@ -96,6 +133,20 @@ export const M365_DEFAULT_OPTION_SETS = [ "flux_v3_progress_messages", "enable_batch_token_processing", "enable_gg_gpt", + "async_client_interaction", + "flux_v3_references", + "flux_v3_references_entities", + "flux_v3_references_ci", + "add_filestore_filetype", + "cwc_code_interpreter_citation_sourceannotations", + "cdxcwc_code_interpreter_hallucinated_url_filter", + "flux_v3_image_gen_enable_dimensions", + "flux_v3_image_gen_enable_non_watermarked_storage", + "flux_v3_image_gen_enable_icon_dimensions", + "flux_v3_image_gen_enable_system_text_with_params", + "flux_v3_image_gen_enable_designer_dimensions_meta_prompting_in_system_prompts", + "flux_v3_image_gen_enable_story", + "rich_responses", ] as const; /** Append the record separator to a JSON-serializable frame. */ @@ -210,6 +261,203 @@ export interface ChatInvocationOptions { * surface omits the key entirely, so it is left out unless set (#10718). */ disconnectBehavior?: string; + /** Client-declared tool plugins (see {@link clientPlugins}); defaults to `[]`. */ + plugins?: JsonRecord[]; + /** OpenAI `tool_choice` echoed to the substrate; defaults to `null`. */ + toolChoice?: unknown; + /** Tool-use nudge sent as `customInstructions` when tools are declared. */ + customInstructions?: string; +} + +/** A client-declared tool in the normalized shape produced by `extractToolSpec`. */ +export interface M365ToolDecl { + name: string; + description: string; + parameters: JsonRecord | null; +} + +/** + * Map normalized OpenAI function tools to the M365 `plugins[]` invocation entries + * (`{Id, Source:"API", Description, Parameters}`), mirroring the community M365 + * convention. Entries without a name are skipped by the extractor upstream. + */ +export function clientPlugins(tools: M365ToolDecl[]): JsonRecord[] { + return tools.map((t) => ({ + Id: t.name, + Source: "API", + Description: t.description, + Parameters: t.parameters ?? {}, + })); +} + +/** True when `toolChoice` permits calling `name` (string / typed / "required"/"auto"). */ +function toolChoiceAllows(toolChoice: unknown, name: string): boolean { + if (toolChoice == null || toolChoice === "auto" || toolChoice === "required") return true; + if (typeof toolChoice === "string") return toolChoice === name; + const fn = (toolChoice as JsonRecord)?.function as JsonRecord | undefined; + return typeof fn?.name === "string" && fn.name === name; +} + +/** A tool call parsed from the model's fenced-block or router output. */ +export interface M365ParsedToolCall { + id: string; + type: string; + name: string; + /** JSON-stringified arguments object, as the OpenAI `tool_calls` shape expects. */ + arguments: string; +} + +const SHELL_TOOL_NAMES = ["bash", "sh", "shell", "powershell", "cmd"] as const; +const FENCED_BLOCK = /```([A-Za-z0-9_-]+)[ \t]*\r?\n([\s\S]*?)\r?\n```/g; + +/** + * Parse the model's fenced-block tool calls out of a completed turn + * (```` ```toolname\n{json args}\n``` ```` — the protocol taught by the prompt). + * Only names the client actually declared are accepted (undeclared names such as + * a hallucinated `unknown_tool` must never reach the caller), and `tool_choice` + * restrictions are enforced the same way. A shell-family block emitted for a + * DECLARED shell tool is normalized into `{command: "..."}`. + */ +export function parseFencedToolCalls( + text: string, + tools: M365ToolDecl[], + toolChoice: unknown +): M365ParsedToolCall[] { + const allowed = new Set(tools.map((t) => t.name)); + const declaredShell = SHELL_TOOL_NAMES.find((n) => allowed.has(n)); + const out: M365ParsedToolCall[] = []; + for (const m of text.matchAll(FENCED_BLOCK)) { + const name = m[1]!; + const body = m[2]!.trim(); + let parsed: unknown; + try { + parsed = JSON.parse(body); + } catch { + parsed = undefined; + } + // Shell-family blocks: keep only for a declared shell tool, normalizing a + // plain-text body (or {"command": ...}) into the canonical arguments object. + if ((SHELL_TOOL_NAMES as readonly string[]).includes(name)) { + const target = allowed.has(name) ? name : declaredShell; + if (!target) continue; + const args = + parsed && typeof parsed === "object" && "command" in (parsed as JsonRecord) + ? (parsed as JsonRecord) + : { command: body }; + out.push({ + id: `call_${crypto.randomUUID()}`, + type: "function", + name: target, + arguments: JSON.stringify(args), + }); + continue; + } + if (!allowed.has(name) || !toolChoiceAllows(toolChoice, name)) continue; + if (parsed == null || typeof parsed !== "object") continue; + out.push({ + id: `call_${crypto.randomUUID()}`, + type: "function", + name, + arguments: JSON.stringify(parsed), + }); + } + return out; +} + +/** A router-turn decision: `decided:false` means the output was unparseable. */ +export interface M365RouterDecision { + decided: boolean; + calls: M365ParsedToolCall[]; +} + +function allowedName(tools: M365ToolDecl[], name: string): boolean { + return tools.some((t) => t.name === name); +} + +function validCall( + name: string, + args: unknown, + tools: M365ToolDecl[], + toolChoice: unknown +): M365ParsedToolCall | null { + if (!name || !allowedName(tools, name) || !toolChoiceAllows(toolChoice, name)) return null; + if (!args || typeof args !== "object") return null; + return { + id: `call_${crypto.randomUUID()}`, + type: "function", + name, + arguments: JSON.stringify(args), + }; +} + +/** + * Parse the router turn's decision (`CALL_TOOL: name({...})` lines / + * `NO_TOOL_NEEDED`), validating every call against the declared tools and + * `tool_choice`. Falls back to the `{"calls":[...]}` JSON envelope. Returns + * `decided:false` when the output is neither shape, so the caller can fall + * through to a plain answer turn instead of guessing. + */ +export function parseToolRouterDecision( + text: string, + tools: M365ToolDecl[], + toolChoice: unknown +): M365RouterDecision { + const trimmed = text.trim(); + const calls: M365ParsedToolCall[] = []; + for (const line of trimmed.split(/\r?\n/)) { + const m = /^CALL_TOOL:\s*(.+)$/i.exec(line.trim()); + if (!m) continue; + const rest = m[1]!; + const start = rest.indexOf("("); + const end = rest.lastIndexOf(")"); + if (start <= 0 || end <= start) continue; + const name = rest.slice(0, start).trim(); + try { + const args = JSON.parse(rest.slice(start + 1, end)); + const call = validCall(name, args, tools, toolChoice); + if (call) calls.push(call); + } catch { + /* malformed JSON on this line — skip */ + } + } + if (calls.length > 0) return { decided: true, calls }; + if (/^no_tool_needed$/i.test(trimmed) || trimmed.toLowerCase().includes("no_tool_needed")) { + return { decided: true, calls: [] }; + } + // Fallback: the {"calls":[{"name","arguments"}]} envelope, optionally fenced. + let probe = trimmed; + const fence = probe.indexOf("```"); + if (fence >= 0) { + probe = probe + .slice(fence + 3) + .replace(/```$/, "") + .trim(); + probe = probe.replace(/^(json|JSON)\s*/, ""); + } + const start = probe.indexOf("{"); + const end = probe.lastIndexOf("}"); + if (start >= 0 && end > start) { + try { + const parsed = JSON.parse(probe.slice(start, end + 1)) as { + calls?: Array<{ name?: unknown; arguments?: unknown }>; + }; + if (Array.isArray(parsed.calls)) { + for (const c of parsed.calls) { + const call = validCall( + typeof c?.name === "string" ? c.name : "", + c?.arguments, + tools, + toolChoice + ); + if (call) calls.push(call); + } + return { decided: true, calls }; + } + } catch { + /* not JSON — undecided */ + } + } + return { decided: false, calls: [] }; } /** @@ -234,12 +482,14 @@ export function resolveChatInvocationOverrides(tier: string | undefined): { } return { optionsSets: [...M365_DEFAULT_OPTION_SETS], - // #10718 — the 2026-08 recapture sends tone:"magic" (lowercase) on the - // individual/EDU surface; the old "" default is part of the dropped shape. - tone: "magic", + // 2026-08-21 capture — the individual/consumer surface now sends "Magic" + // (capitalized), matching the enterprise tone literal. The #10718 + // lowercase "magic" is part of the shape that gets silently dropped. + tone: "Magic", allowedMessageTypes: ALLOWED_MESSAGE_TYPES, - // Omitted entirely on the individual/EDU wire (see ChatInvocationOptions). - disconnectBehavior: undefined, + // 2026-08-21 capture — disconnectBehavior:"continue" is now present on the + // individual/consumer wire too, not just enterprise (see ChatInvocationOptions). + disconnectBehavior: "continue", }; } @@ -268,16 +518,33 @@ export function resolveToneForModel(model: string | undefined): string | undefin /** * Build the `type:4` chat invocation frame body (not yet `\x1e`-terminated). - * Mirrors the argument shape recaptured from a working `m365.cloud.microsoft/chat` - * client in 2026-08 (#10718). Notable differences from the pre-#10718 shape: a - * populated `clientInfo` + `productThreadType:"Office"`, a `conversationId` - * matching the WS URL query, a rich `message` object, and no - * `spokenTextMode` / `extraExtensionParameters` / `isSbsSupported` / - * `renderReferencesBehindEOS` / `disconnectBehavior` — none of those are still - * observed on the wire, and the stale shape gets closed immediately after the - * invocation. + * Base shape from the #10718 recapture (populated `clientInfo` + + * `productThreadType:"Office"`, a `conversationId` matching the WS URL query, a + * rich `message` object), extended per a 2026-08-21 live capture that found the + * #10718 shape alone no longer produces a `type:1 target:"update"` frame — the + * socket only replies with keepalive pings and closes. The additions below + * (richer `clientInfo`, non-empty `plugins`, `extraExtensionParameters`, + * `isSbsSupported`, `renderReferencesBehindEOS`, + * `message.connectedFederatedConnections`, and `disconnectBehavior` on every + * tier) are exactly the fields the 2026-08-21 capture had that this shape was + * missing; the #10718 fields (`conversationId`, `productThreadType`, + * `toolChoice`, `message.attachments`) are kept as-is since removing them was + * not verified against a live socket. */ export function buildChatInvocation(opts: ChatInvocationOptions): Record { + const clientInfo = { + clientAppName: "Office", + clientPlatform: "mcmcopilot-web", + clientEntrypoint: "mcmcopilot-officeweb", + clientSessionId: opts.sessionId, + ProductCategory: "Chat", + clientAppType: "Web", + productEntryPoint: "ChatPanel", + deviceOS: "Windows", + deviceType: "Desktop", + clientPlatformVersion: "10", + }; + return { type: 4, target: "chat", @@ -288,17 +555,17 @@ export function buildChatInvocation(opts: ChatInvocationOptions): Record | null): boolea return !!frame && frame.type === 3; } +/** + * Extract the error message from a `type:3` completion frame that carries one + * (`frame.error.message` / `frame.error`). A clean completion returns null — + * without this check a server-side invocation error surfaces as a silent empty + * `stop`, indistinguishable from a genuine empty reply. + */ +export function extractCompletionError(frame: Record | null): string | null { + if (!frame || frame.type !== 3) return null; + const error = frame.error; + if (!error || typeof error !== "object") return null; + const message = (error as JsonRecord).message; + return typeof message === "string" && message.length > 0 ? message : JSON.stringify(error); +} + +/** + * True for messages that carry tool/search/code PROGRESS rather than answer text + * (`messageType:"Progress"`, or the SearchResults/Code/ToolCall content types). + * Such text must never be folded into the streamed answer. + */ +function isToolProgressMessage(m: Record): boolean { + if (m.messageType === "Progress") return true; + const ct = m.contentType; + return ct === "SearchResults" || ct === "Code" || ct === "ToolCall" || ct === "EarlyProgress"; +} + +/** + * True when an update frame is a tool-progress frame — it carries Progress / + * SearchResults / Code / ToolCall messages alongside (possibly) a `writeAtCursor` + * increment that belongs to that progress, not to the answer (the browser client + * suppresses such writeAtCursor deltas; so must we). + */ +export function isToolProgressFrame(frame: Record | null): boolean { + if (!isUpdateFrame(frame)) return false; + const args = frame.arguments; + const first = Array.isArray(args) ? (args[0] as Record | undefined) : undefined; + const messages = first?.messages; + if (!Array.isArray(messages)) return false; + return messages.some( + (m) => !!m && typeof m === "object" && isToolProgressMessage(m as Record) + ); +} + /** True when an update frame is flagged as the last update of the turn. */ export function isLastUpdate(frame: Record | null): boolean { if (!isUpdateFrame(frame)) return false; @@ -366,7 +681,7 @@ export function extractBotText(frame: Record | null): string | if (!m) continue; const author = m.author; const text = m.text; - if (m.messageType === "Progress" || m.contentType === "EarlyProgress") continue; + if (isToolProgressMessage(m)) continue; if ((author === "bot" || author === undefined) && typeof text === "string" && text.length > 0) { return text; } @@ -425,6 +740,9 @@ export function accumulateBotContent( previous: string, frame: Record | null ): { delta: string; next: string } { + // A tool-progress frame's writeAtCursor belongs to the progress card (search + // queries, code interpreter output…), not to the answer text. + if (isToolProgressFrame(frame)) return { delta: "", next: previous }; const snapshot = extractBotText(frame); if (snapshot) { return { delta: incrementalDelta(previous, snapshot), next: snapshot }; diff --git a/open-sse/executors/copilot-m365-web.ts b/open-sse/executors/copilot-m365-web.ts index 5bdaf9ae9bf..4bfe4014b44 100644 --- a/open-sse/executors/copilot-m365-web.ts +++ b/open-sse/executors/copilot-m365-web.ts @@ -4,10 +4,13 @@ import { sanitizeErrorMessage } from "../utils/error.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorLog } from "./base.ts"; import { buildPrompt, + buildRouterPrompt, buildWsUrl, currentM365AccessToken, currentM365ChathubPath, decodeJwtClaims, + extractToolSpec, + flattenMessages, redactWsUrl, refreshM365AccessToken, resolveConnectionParams, @@ -16,20 +19,30 @@ import { import { accumulateBotContent, buildChatInvocation, + clientPlugins, encodeFrame, + extractCompletionError, extractFinalResultMessage, handshakeError, handshakeFrame, isCompletionFrame, isUpdateFrame, + keepaliveFrame, metricsFrame, + parseFencedToolCalls, parseFrame, + parseToolRouterDecision, resolveChatInvocationOverrides, resolveToneForModel, splitFrames, } from "./copilot-m365-frames.ts"; type JsonRecord = Record; +type M365ToolDecl = { + name: string; + description: string; + parameters: JsonRecord | null; +}; let WebSocketCtor: typeof WebSocket = WebSocket; export function __setCopilotM365WebSocketForTesting(ctor: typeof WebSocket): () => void { @@ -70,6 +83,97 @@ function errorResponse(message: string, status = 502): Response { }); } +/** Consume one wsChat SSE stream to its full text (router turns are read fully). */ +async function readSseText(stream: ReadableStream): Promise { + const reader = stream.getReader(); + const decoder = new TextDecoder(); + let fullText = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + for (const line of decoder.decode(value, { stream: true }).split("\n")) { + if (!line.startsWith("data: ")) continue; + const data = line.slice(6).trim(); + if (!data || data === "[DONE]") continue; + try { + const parsed = JSON.parse(data) as JsonRecord; + const choices = parsed.choices; + const choice = (Array.isArray(choices) ? choices[0] : undefined) as + { delta?: { content?: unknown } } | undefined; + if (typeof choice?.delta?.content === "string") fullText += choice.delta.content; + } catch { + /* skip malformed SSE lines */ + } + } + } + return fullText; +} + +/** Build the tool_calls result for a routed decision (stream + non-stream). */ +function toolCallsResult( + calls: Array<{ id: string; type: string; name: string; arguments: string }>, + opts: { stream: boolean; model: string; wsUrl: string } +) { + if (opts.stream) { + let sse = sseChunk(opts.model, { role: "assistant", content: null }); + for (let i = 0; i < calls.length; i++) { + sse += sseChunk(opts.model, { + tool_calls: [ + { + index: i, + id: calls[i]!.id, + type: calls[i]!.type, + function: { name: calls[i]!.name, arguments: calls[i]!.arguments }, + }, + ], + }); + } + sse += sseChunk(opts.model, {}, "tool_calls") + "data: [DONE]\n\n"; + return { + response: new Response(sse, { + headers: { + "Content-Type": "text/event-stream", + "Cache-Control": "no-cache", + Connection: "keep-alive", + }, + }), + url: redactWsUrl(opts.wsUrl), + headers: {}, + transformedBody: { model: opts.model, toolCalls: calls.length }, + }; + } + return { + response: new Response( + JSON.stringify({ + id: `chatcmpl-copilot-m365-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model: opts.model, + choices: [ + { + index: 0, + message: { + role: "assistant", + content: null, + tool_calls: calls.map((c) => ({ + id: c.id, + type: c.type, + function: { name: c.name, arguments: c.arguments }, + })), + }, + finish_reason: "tool_calls", + }, + ], + usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }, + }), + { headers: { "Content-Type": "application/json" } } + ), + url: redactWsUrl(opts.wsUrl), + headers: {}, + transformedBody: { model: opts.model, toolCalls: calls.length }, + }; +} + export class CopilotM365WebExecutor extends BaseExecutor { constructor() { super("copilot-m365-web", { id: "copilot-m365-web", baseUrl: "wss://substrate.office.com" }); @@ -80,12 +184,15 @@ export class CopilotM365WebExecutor extends BaseExecutor { prompt: string; model: string; tier?: string; + tools?: M365ToolDecl[]; + toolChoice?: unknown; signal?: AbortSignal; log?: ExecutorLog | null; }): Promise> { // #6210 — observability for the empty-response class. The access_token rides // in the WS query string, so every URL logged here goes through redactWsUrl(). const log = input.log ?? null; + const toolMode = (input.tools?.length ?? 0) > 0; return new ReadableStream( { start: async (controller) => { @@ -94,6 +201,11 @@ export class CopilotM365WebExecutor extends BaseExecutor { let settled = false; let buffer = ""; let previousText = ""; + // Tool-call streaming: with tools declared, content is emitted with a + // small tail holdback until a fenced block opens — from then on everything + // is buffered and resolved into `tool_calls` at finish, never as content. + let pendingTail = ""; + let fenceSeen = false; let finalResultMessage = ""; let handshakeComplete = false; @@ -112,17 +224,50 @@ export class CopilotM365WebExecutor extends BaseExecutor { if (settled) return; settled = true; cleanup(); - // Last-resort fallback (#6210): some EDU turns surface the answer only in the - // type:2 invocation result. Emit it if nothing was streamed. + // Last-resort fallback (#6210): some EDU turns surface the answer only + // in the type:2 invocation result. Treat it as the turn text. if (!previousText && finalResultMessage) { + previousText = finalResultMessage; + } + // Tool-call resolution: parse the fenced-block protocol out of the + // completed turn and, when the model called declared tools, close the + // stream with OpenAI `tool_calls` instead of plain content. + const calls = toolMode + ? parseFencedToolCalls(previousText, input.tools ?? [], input.toolChoice) + : []; + if (calls.length > 0) { controller.enqueue( - encoder.encode(sseChunk(input.model, { content: finalResultMessage })) + encoder.encode(sseChunk(input.model, { role: "assistant", content: null })) ); - } else if (!previousText && !finalResultMessage) { + for (let i = 0; i < calls.length; i++) { + const call = calls[i]!; + controller.enqueue( + encoder.encode( + sseChunk(input.model, { + tool_calls: [ + { + index: i, + id: call.id, + type: call.type, + function: { name: call.name, arguments: call.arguments }, + }, + ], + }) + ) + ); + } + controller.enqueue(encoder.encode(sseChunk(input.model, {}, "tool_calls"))); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + return; + } + if (!previousText) { // #7858 — a turn that completed with no content in ANY known shape is // indistinguishable, from the outside, from a genuine successful-but-empty // reply. Fail loudly instead of a silent `stop`, per Hard Rule #12. - const tierNote = input.tier ? `resolved tier: ${input.tier}` : "resolved tier: individual (default)"; + const tierNote = input.tier + ? `resolved tier: ${input.tier}` + : "resolved tier: individual (default)"; const message = sanitizeErrorMessage( `Microsoft 365 Copilot turn completed with no content in any known frame ` + `shape (${tierNote}). Possible causes: an unrecognized frame shape for ` + @@ -134,6 +279,11 @@ export class CopilotM365WebExecutor extends BaseExecutor { controller.close(); return; } + // No tool calls: flush any holdback tail as ordinary content and stop. + if (pendingTail) { + controller.enqueue(encoder.encode(sseChunk(input.model, { content: pendingTail }))); + pendingTail = ""; + } controller.enqueue(encoder.encode(sseChunk(input.model, {}, "stop"))); controller.enqueue(encoder.encode("data: [DONE]\n\n")); controller.close(); @@ -144,7 +294,9 @@ export class CopilotM365WebExecutor extends BaseExecutor { settled = true; cleanup(); const message = sanitizeErrorMessage(reason); - controller.enqueue(encoder.encode(`data: ${JSON.stringify({ error: { message } })}\n\n`)); + controller.enqueue( + encoder.encode(`data: ${JSON.stringify({ error: { message } })}\n\n`) + ); controller.close(); }; @@ -194,6 +346,18 @@ export class CopilotM365WebExecutor extends BaseExecutor { isStartOfSession: true, ...overrides, tone, + // Declare the client's tools natively too (plugins + toolChoice + + // a customInstructions nudge); the fenced-block protocol in the + // prompt remains the parseable path. + ...(toolMode + ? { + plugins: clientPlugins(input.tools ?? []), + toolChoice: input.toolChoice ?? null, + customInstructions: + "You have access to real tools provided by the calling application. " + + "Call tools directly when needed. Do not say tools are unavailable.", + } + : {}), }) ); // #10718 — the invocation and its type:1 Metrics follow-up must land @@ -233,6 +397,17 @@ export class CopilotM365WebExecutor extends BaseExecutor { continue; } + // SignalR keepalive: the server pings with type:6 and expects the + // exact echo back, or it drops the socket mid-turn on long agentic runs. + if (frame?.type === 6) { + try { + ws?.send(keepaliveFrame()); + } catch { + /* socket already closing — the close handler finishes the stream */ + } + continue; + } + const { delta, next } = accumulateBotContent(previousText, frame); if (!delta && next === previousText) { // #7858 AC2/AC3 — log unrecognized-shape update frames by KEY only, so @@ -243,7 +418,25 @@ export class CopilotM365WebExecutor extends BaseExecutor { } previousText = next; if (delta) { - controller.enqueue(encoder.encode(sseChunk(input.model, { content: delta }))); + if (!toolMode) { + controller.enqueue(encoder.encode(sseChunk(input.model, { content: delta }))); + } else if (!fenceSeen) { + // Hold back a 12-char tail so a "```" opener straddling a chunk + // boundary is never emitted as content; once any fence opens, + // buffer everything for the finish-time tool-call resolution. + pendingTail += delta; + if (pendingTail.includes("```")) { + fenceSeen = true; + } else if (pendingTail.length > 12) { + const cut = pendingTail.length - 12; + controller.enqueue( + encoder.encode( + sseChunk(input.model, { content: pendingTail.slice(0, cut) }) + ) + ); + pendingTail = pendingTail.slice(cut); + } + } } const finalMsg = extractFinalResultMessage(frame); @@ -251,6 +444,16 @@ export class CopilotM365WebExecutor extends BaseExecutor { finalResultMessage = finalMsg; } + // A type:3 carrying an error is a FAILED turn; without this it + // would finish() into a silent empty stop. + const completionError = extractCompletionError(frame); + if (completionError) { + clearTimeout(timeout); + log?.debug?.("M365_WS", `completion error: ${completionError}`); + abort(`Microsoft 365 Copilot invocation failed: ${completionError}`); + return; + } + if (isCompletionFrame(frame)) { clearTimeout(timeout); finish(); @@ -310,8 +513,7 @@ export class CopilotM365WebExecutor extends BaseExecutor { const current = currentM365AccessToken(credentials); if (current && !tokenNeedsRefresh(current)) return; - const tid = - decodeJwtClaims(current)?.tid || (typeof psd.tid === "string" ? psd.tid : "") || ""; + const tid = decodeJwtClaims(current)?.tid || (typeof psd.tid === "string" ? psd.tid : "") || ""; const result = await refreshM365AccessToken(refreshToken, tid, log ?? undefined); if ("error" in result) { // Fall through with the existing token — the WS layer will surface the failure. @@ -355,7 +557,19 @@ export class CopilotM365WebExecutor extends BaseExecutor { const body = input.body as JsonRecord | undefined; const model = input.model || (body?.model as string) || "copilot-m365"; const stream = input.stream !== false; - const prompt = buildPrompt(body).trim(); + const { tools, toolChoice } = extractToolSpec(body); + const routerActive = tools.length > 0 && toolChoice !== "none"; + // Router planning: the router turn decides tool use; the answer turn (when the + // router selects none) must be RE-FRAMED as an answer request — a raw history + // continuation makes the model keep emitting the router's decision format. + const flat = flattenMessages(body); + const prompt = ( + routerActive + ? "Please answer the following request in full, using the tool results already " + + "provided in the conversation. Do not output tool-routing decisions.\n\n" + + flat + : buildPrompt(body) + ).trim(); if (!prompt) { return { @@ -383,13 +597,44 @@ export class CopilotM365WebExecutor extends BaseExecutor { } const wsUrl = buildWsUrl(connectionParams); + let answerWsUrl: string | null = null; try { + // Router planning turn — ask the model as a tool-SELECTION assistant. Asking + // it to "use" a client tool gets refused (it checks its own plugin registry); + // printing a routing decision as text bypasses that refusal. + if (routerActive) { + const routerStream = await this.wsChat({ + wsUrl, + prompt: buildRouterPrompt(flat, tools, toolChoice), + model, + tier: connectionParams.tier, + signal: input.signal ?? undefined, + log: input.log, + }); + const routerText = await readSseText(routerStream); + const decision = parseToolRouterDecision(routerText, tools, toolChoice); + input.log?.debug?.( + "M365_TOOLS", + `router decided=${decision.decided} calls=${decision.calls.length}` + ); + if (decision.decided && decision.calls.length > 0) { + return toolCallsResult(decision.calls, { stream, model, wsUrl }); + } + // No tool needed (or unparseable): answer in a FRESH conversation below. + // Reusing the router's ConversationId makes the answer turn a continuation + // of the routing dialog, and the model keeps emitting the router's decision + // format (NO_TOOL_NEEDED) as the answer. + answerWsUrl = buildWsUrl(connectionParams); + } + const wsStream = await this.wsChat({ - wsUrl, + wsUrl: answerWsUrl ?? wsUrl, prompt, model, tier: connectionParams.tier, + tools, + toolChoice, signal: input.signal ?? undefined, log: input.log, }); @@ -412,6 +657,12 @@ export class CopilotM365WebExecutor extends BaseExecutor { const reader = wsStream.getReader(); const decoder = new TextDecoder(); let fullText = ""; + const toolCalls: Array<{ + id: string; + type: string; + name: string; + arguments: string; + }> = []; while (true) { const { done, value } = await reader.read(); if (done) break; @@ -421,14 +672,60 @@ export class CopilotM365WebExecutor extends BaseExecutor { if (!data || data === "[DONE]") continue; try { const parsed = JSON.parse(data); - const content = parsed.choices?.[0]?.delta?.content; + const choice = parsed.choices?.[0]; + const content = choice?.delta?.content; if (typeof content === "string") fullText += content; + for (const tc of choice?.delta?.tool_calls ?? []) { + toolCalls.push({ + id: String(tc.id ?? ""), + type: String(tc.type ?? "function"), + name: String(tc.function?.name ?? ""), + arguments: String(tc.function?.arguments ?? "{}"), + }); + } } catch { /* skip malformed SSE lines */ } } } + // Tool-call turn: content stops at the first fence, the calls ride in + // `tool_calls` with finish_reason "tool_calls" (OpenAI agentic-loop shape). + if (toolCalls.length > 0) { + const fenceIndex = fullText.indexOf("```"); + const content = fenceIndex > 0 ? fullText.slice(0, fenceIndex).trim() : null; + return { + response: new Response( + JSON.stringify({ + id: `chatcmpl-copilot-m365-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model, + choices: [ + { + index: 0, + message: { + role: "assistant", + content, + tool_calls: toolCalls.map((c) => ({ + id: c.id, + type: c.type, + function: { name: c.name, arguments: c.arguments }, + })), + }, + finish_reason: "tool_calls", + }, + ], + usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }, + }), + { headers: { "Content-Type": "application/json" } } + ), + url: redactWsUrl(answerWsUrl ?? wsUrl), + headers: {}, + transformedBody: { model, toolCalls: toolCalls.length }, + }; + } + return { response: new Response( JSON.stringify({ diff --git a/open-sse/executors/cursor.ts b/open-sse/executors/cursor.ts index ebcfe053bd2..8ffca7b318a 100644 --- a/open-sse/executors/cursor.ts +++ b/open-sse/executors/cursor.ts @@ -12,6 +12,7 @@ declare const EdgeRuntime: string | undefined; import { BaseExecutor, mergeUpstreamExtraHeaders } from "./base.ts"; import { PROVIDERS, HTTP_STATUS } from "../config/constants.ts"; +import { getAccessToken } from "../services/tokenRefresh.ts"; import { buildAgentRequestBody, decodeAgentServerMessage, @@ -83,6 +84,12 @@ import { composerReasoningRemainder, } from "./cursor/composer.ts"; import { CursorServerConfigError, resolveCursorAgentUrl } from "./cursor/agentEndpoint.ts"; +import { + classifyCursorError, + isCursorBenignCancelError, + resolveCursorEmptyTurnError, + type ClassifiedCursorError, +} from "./cursor/cursorErrors.ts"; import { getActiveSyncedCatalog } from "../../src/lib/db/models/activeSyncedCatalog.ts"; // Composer helpers re-exported for external importers (tests). export { @@ -250,19 +257,33 @@ function tryParseJsonError(payload: Buffer): { message: string; status: number } if (!text.includes('"error"')) return null; const parsed = JSON.parse(text); const err = parsed?.error || {}; - const message = + const rawMessage = err?.details?.[0]?.debug?.details?.title || err?.details?.[0]?.debug?.details?.detail || err?.message || - text; - const status = - err?.code === "resource_exhausted" ? HTTP_STATUS.RATE_LIMITED : HTTP_STATUS.BAD_REQUEST; - return { message, status }; + (typeof err?.code === "string" ? `${err.code}: ${text}` : text); + const codeHint = + typeof err?.code === "string" && + !String(rawMessage).toLowerCase().includes(err.code.toLowerCase()) + ? `${err.code}: ${rawMessage}` + : String(rawMessage); + const classified = classifyCursorError(codeHint); + return { message: classified.message, status: classified.status }; } catch { return null; } } +/** True when the turn produced no client-visible assistant payload. */ +function isCursorEmptyTurn(ctx: StreamCtx): boolean { + return ( + ctx.totalText.length === 0 && + ctx.thinkingText.length === 0 && + ctx.toolCalls.length === 0 && + !ctx.composerInlineToolCallsEmitted + ); +} + // ─── Phase 4: streaming dispatch context ─────────────────────────────────── // // One StreamCtx flows through a single execute() call. It owns the live @@ -355,6 +376,27 @@ function emitChunk(ctx: StreamCtx, delta: object, finishReason: string | null = ctx.emit(`data: ${JSON.stringify(payload)}\n\n`); } +/** + * Emit a terminal OpenAI SSE error matching `buildStreamErrorChunks` shape + * (`finish_reason: "error"` + `error.message`) so #8649 sawError stands down + * and Model Test All keeps the classified Cursor message. + */ +export function emitCursorSseError(ctx: StreamCtx, classified: ClassifiedCursorError): void { + const payload = { + id: ctx.responseId, + object: "chat.completion.chunk", + created: ctx.created, + model: ctx.model, + choices: [{ index: 0, delta: {}, finish_reason: "error" }], + error: { + message: classified.message, + type: classified.type, + }, + }; + ctx.emit(`data: ${JSON.stringify(payload)}\n\n`); + ctx.emit("data: [DONE]\n\n"); +} + export function buildCursorUsage(ctx: StreamCtx, body: { messages?: ChatMessage[] }) { const promptTokens = estimateInputTokens(body); const completionTokens = @@ -1441,6 +1483,17 @@ export class CursorExecutor extends BaseExecutor { finishLifecycle(ctx, false); controller.close(); } catch (err) { + // OpenCodex: NGHTTP2_CANCEL after client-tool suspend is expected — finish + // the SSE turn instead of surfacing a transport failure. + if ( + isCursorBenignCancelError(err) && + (ctx.totalText.length > 0 || ctx.pendingToolCalls.size > 0) + ) { + this.finalizeSseStream(ctx, body); + finishLifecycle(ctx, false); + controller.close(); + return; + } finishLifecycle(ctx, true); controller.error(err); } @@ -1468,10 +1521,23 @@ export class CursorExecutor extends BaseExecutor { try { await this.driveH2(h2, ctx, mcpTools, blobStore, clientPlatform, todoHistory, signal); } catch (err) { + if ( + isCursorBenignCancelError(err) && + (ctx.totalText.length > 0 || ctx.pendingToolCalls.size > 0) + ) { + finishLifecycle(ctx, false); + return { + response: this.buildResponseFromCtx(ctx, body), + url, + headers, + transformedBody: body, + }; + } finishLifecycle(ctx, true); const message = err instanceof Error ? err.message : String(err); + const classified = classifyCursorError(message); return { - response: buildErrorResponse(HTTP_STATUS.SERVER_ERROR, message, "connection_error"), + response: buildErrorResponse(classified.status, classified.message, classified.type), url, headers, transformedBody: body, @@ -1493,24 +1559,22 @@ export class CursorExecutor extends BaseExecutor { */ private finalizeSseStream(ctx: StreamCtx, body: { messages?: ChatMessage[] }) { if (ctx.midStreamError && ctx.totalText.length === 0) { - const payload = { - id: ctx.responseId, - object: "chat.completion.chunk", - created: ctx.created, - model: ctx.model, - choices: [], - error: { - message: ctx.midStreamError.message, - type: - ctx.midStreamError.status === HTTP_STATUS.RATE_LIMITED - ? "rate_limit_error" - : "api_error", - }, - }; - ctx.emit(`data: ${JSON.stringify(payload)}\n\n`); - ctx.emit("data: [DONE]\n\n"); + emitCursorSseError(ctx, classifyCursorError(ctx.midStreamError.message)); return; } + + // Silent empty turn (auth accepted, no text) — surface actionable error instead of + // an empty assistant completion that chatCore maps to opaque "empty content" 502. + if (isCursorEmptyTurn(ctx) && ctx.endReason && ctx.endReason !== "tool_calls") { + emitCursorSseError( + ctx, + resolveCursorEmptyTurnError({ + upstreamMessage: ctx.midStreamError?.message, + }) + ); + return; + } + if (!ctx.emittedRoleChunk) { // Edge case: empty response. Emit a role chunk so clients see at least // one delta before finish. @@ -1565,18 +1629,34 @@ export class CursorExecutor extends BaseExecutor { */ private buildResponseFromCtx(ctx: StreamCtx, body: { messages?: ChatMessage[] }): Response { if (ctx.midStreamError && ctx.totalText.length === 0) { + const classified = classifyCursorError(ctx.midStreamError.message); + return new Response( + JSON.stringify({ + error: { + message: classified.message, + type: classified.type, + }, + }), + { + status: classified.status, + headers: { "Content-Type": "application/json" }, + } + ); + } + + if (isCursorEmptyTurn(ctx) && ctx.endReason && ctx.endReason !== "tool_calls") { + const empty = resolveCursorEmptyTurnError({ + upstreamMessage: ctx.midStreamError?.message, + }); return new Response( JSON.stringify({ error: { - message: ctx.midStreamError.message, - type: - ctx.midStreamError.status === HTTP_STATUS.RATE_LIMITED - ? "rate_limit_error" - : "api_error", + message: empty.message, + type: empty.type, }, }), { - status: ctx.midStreamError.status, + status: empty.status, headers: { "Content-Type": "application/json" }, } ); @@ -1655,8 +1735,23 @@ export class CursorExecutor extends BaseExecutor { ); } - async refreshCredentials() { - return null; + async refreshCredentials(credentials, log) { + if (!credentials?.refreshToken) { + log?.warn?.( + "TOKEN_REFRESH", + "Cursor: no refresh token available, re-authentication required" + ); + return null; + } + const result = await getAccessToken("cursor", credentials, log); + if (!result || result.error) { + log?.warn?.( + "TOKEN_REFRESH", + `Cursor: token refresh failed${result?.error ? ` (${result.error})` : ""} — re-authentication required` + ); + return null; + } + return result; } } diff --git a/open-sse/executors/cursor/cursorErrors.ts b/open-sse/executors/cursor/cursorErrors.ts new file mode 100644 index 00000000000..9d05eaab6d3 --- /dev/null +++ b/open-sse/executors/cursor/cursorErrors.ts @@ -0,0 +1,269 @@ +/** + * Classify Cursor transport / Connect / gRPC error text into actionable categories. + * Modeled on OpenCodex `adapters/cursor/cursor-errors.ts` (safe messages + quota vs size). + */ + +const ABSOLUTE_PATH_PATTERN = + /(?:\/Users\/[^ "';,]+|\/home\/[^ "';,]+|[A-Za-z]:\\Users\\[^ "';,]+)/g; +const CURSOR_CREDENTIAL_PATTERN = + /\b(authorization|auth[_-]?token|cursor[_-]?token|bearer)=([^&\s"',;]+)/gi; + +const QUOTA_RATE_CUES = [ + "too many requests", + "quota", + "rate limit", + "rate-limit", + "throttl", + "out of usage", + "increase limits", + "actionrequired", +]; +const REQUEST_TOO_LARGE_PATTERNS: (string | RegExp)[] = [ + "tool catalog too large", + "tool registration too large", + "too many tools", + "message too large", + "payload too large", + "request too large", + /request exceeds .*size/, + /request (?:body|size) exceeds .*(?:size|limit)/, + "maximum allowed size", +]; + +export type CursorErrorKind = + "rate_limit" | "auth" | "invalid" | "overload" | "timeout" | "connection" | "upstream"; + +export type ClassifiedCursorError = { + kind: CursorErrorKind; + /** HTTP status to surface to OmniRoute clients. */ + status: number; + /** OpenAI-style error.type */ + type: string; + /** Secret-safe user-facing message with category prefix. */ + message: string; +}; + +function sanitize(value: string): string { + return value + .replace(CURSOR_CREDENTIAL_PATTERN, "$1=[REDACTED]") + .replace(ABSOLUTE_PATH_PATTERN, "[REDACTED_PATH]") + .replace(/eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+/g, "[REDACTED_JWT]"); +} + +export function isCursorRequestTooLargeDetail(lowerMessage: string): boolean { + if (QUOTA_RATE_CUES.some((cue) => lowerMessage.includes(cue))) return false; + return REQUEST_TOO_LARGE_PATTERNS.some((pattern) => + typeof pattern === "string" ? lowerMessage.includes(pattern) : pattern.test(lowerMessage) + ); +} + +function errorMessage(value: unknown): string { + if (value instanceof Error) return value.message; + if (typeof value === "string") return value; + return String(value ?? ""); +} + +function errorCode(value: unknown): string { + if (typeof value !== "object" || !value || !("code" in value)) return ""; + const code = (value as { code?: unknown }).code; + return code === undefined || code === null ? "" : String(code); +} + +/** + * True when Cursor intentionally cancelled the HTTP/2 stream after a client-tool + * suspend (OpenCodex `isCursorBenignCancelError`). Not an upstream failure. + */ +export function isCursorBenignCancelError(value: unknown): boolean { + const message = errorMessage(value).toLowerCase(); + const code = errorCode(value).toUpperCase(); + if (code === "NGHTTP2_CANCEL") return true; + if (message.includes("nghttp2_cancel")) return true; + if (message.includes("cursor stream suspended")) return true; + return false; +} + +export function classifyCursorErrorKind(rawMessage: string): CursorErrorKind { + const lower = rawMessage.toLowerCase(); + + if (lower.includes("resource_exhausted") || lower.includes("resource exhausted")) { + return isCursorRequestTooLargeDetail(lower) ? "invalid" : "rate_limit"; + } + if (QUOTA_RATE_CUES.some((cue) => lower.includes(cue))) return "rate_limit"; + + // Live Cursor out-of-usage for premium models often surfaces as: + // not_found: AI Model Not Found (reset after 109h …) + // OmniRoute may also append "(reset after …)" after classification; treat the + // Cursor-specific "AI Model Not Found" cue as rate/quota either way. + if ( + lower.includes("ai model not found") || + (lower.includes("reset after") && lower.includes("model not found")) + ) { + return "rate_limit"; + } + + if ( + lower.includes("unauthenticated") || + lower.includes("unauthorized") || + lower.includes("permission_denied") || + lower.includes("permission denied") || + lower.includes("forbidden") || + lower.includes("invalid token") || + lower.includes("expired token") || + lower.includes("authentication") || + lower.includes("access denied") + ) { + return "auth"; + } + + if ( + lower.includes("unavailable") || + lower.includes("overloaded") || + lower.includes("temporarily") || + lower.includes("server is busy") + ) { + return "overload"; + } + + if ( + lower.includes("invalid") || + lower.includes("not found") || + lower.includes("unsupported") || + lower.includes("malformed") || + lower.includes("unimplemented") + ) { + return "invalid"; + } + + if ( + lower.includes("timed out") || + lower.includes("timeout") || + lower.includes("etimedout") || + lower.includes("deadline") + ) { + return "timeout"; + } + + if ( + lower.includes("econnreset") || + lower.includes("econnrefused") || + lower.includes("goaway") || + lower.includes("nghttp2") || + lower.includes("socket hang up") || + lower.includes("connection reset") + ) { + return "connection"; + } + + return "upstream"; +} + +function kindToStatus(kind: CursorErrorKind): number { + switch (kind) { + case "rate_limit": + return 429; + case "auth": + return 401; + case "invalid": + return 400; + case "overload": + case "timeout": + case "connection": + case "upstream": + default: + return 502; + } +} + +function kindToType(kind: CursorErrorKind): string { + switch (kind) { + case "rate_limit": + return "rate_limit_error"; + case "auth": + return "authentication_error"; + case "invalid": + return "invalid_request_error"; + default: + return "api_error"; + } +} + +function kindPrefix(kind: CursorErrorKind): string { + switch (kind) { + case "rate_limit": + return "Cursor rate limit / usage exceeded"; + case "auth": + return "Cursor authentication failed"; + case "invalid": + return "Cursor invalid request"; + case "overload": + return "Cursor server overloaded"; + case "timeout": + return "Cursor request timed out"; + case "connection": + return "Cursor connection failed"; + default: + return "Cursor upstream error"; + } +} + +/** Produce a classified, secret-safe Cursor error for HTTP / SSE responses. */ +export function classifyCursorError(rawMessage: string): ClassifiedCursorError { + const kind = classifyCursorErrorKind(rawMessage); + const detail = sanitize(rawMessage) + .replace(/resource[_ ]exhausted/gi, "resource limit exceeded") + .slice(0, 500); + const prefix = kindPrefix(kind); + const message = detail.startsWith(prefix) ? detail : detail ? `${prefix}: ${detail}` : prefix; + return { + kind, + status: kindToStatus(kind), + type: kindToType(kind), + message, + }; +} + +export const CURSOR_EMPTY_TURN_MESSAGE = + 'Cursor returned an empty turn (often usage/quota exhausted). Try model "auto", or check Usage → Provider Limits / raise Cursor limits.'; + +/** + * Resolve the error to emit when a Cursor turn ends with no assistant text/tool_calls. + * Prefer classifying an upstream JSON/error message; otherwise use the empty-turn hint. + * When `quotaExhaustedHint` is true (fresh Provider Limits cache), force 429. + */ +export function resolveCursorEmptyTurnError(options: { + upstreamMessage?: string | null; + quotaExhaustedHint?: boolean; +}): ClassifiedCursorError { + const upstream = options.upstreamMessage?.trim(); + if (upstream) { + const classified = classifyCursorError(upstream); + if (options.quotaExhaustedHint && classified.kind !== "auth") { + return { + ...classified, + kind: "rate_limit", + status: 429, + type: "rate_limit_error", + message: classified.message.includes("usage") + ? classified.message + : `${classified.message} (${CURSOR_EMPTY_TURN_MESSAGE})`, + }; + } + return classified; + } + + if (options.quotaExhaustedHint) { + return { + kind: "rate_limit", + status: 429, + type: "rate_limit_error", + message: CURSOR_EMPTY_TURN_MESSAGE, + }; + } + + return { + kind: "upstream", + status: 502, + type: "api_error", + message: CURSOR_EMPTY_TURN_MESSAGE, + }; +} diff --git a/open-sse/executors/forceResponsesUpstream.ts b/open-sse/executors/forceResponsesUpstream.ts index 0c545d980e6..4de8961a3b5 100644 --- a/open-sse/executors/forceResponsesUpstream.ts +++ b/open-sse/executors/forceResponsesUpstream.ts @@ -28,6 +28,17 @@ export function shouldForceResponsesUpstream( const providerSpecificData = credentials?.providerSpecificData ?? null; if (providerSpecificData?._omnirouteForceResponsesUpstream === true) return true; if (getOpenAICompatibleType(provider, providerSpecificData) === "responses") return false; + // apiType="chat" means the operator explicitly chose the chat/completions + // wire. Don't second-guess that choice by forcing /responses just because the + // body carries namespace tools — the standard namespace→flatten path + // (openai-responses.ts) handles those correctly for chat backends. + if ( + providerSpecificData && + typeof providerSpecificData.apiType === "string" && + providerSpecificData.apiType === "chat" + ) { + return false; + } const hasResponsesShape = body.input !== undefined || diff --git a/open-sse/executors/freebuff.ts b/open-sse/executors/freebuff.ts index bc2de30f632..f15b3e430ef 100644 --- a/open-sse/executors/freebuff.ts +++ b/open-sse/executors/freebuff.ts @@ -1,7 +1,6 @@ -import { - BaseExecutor, - type ExecuteInput, -} from "./base.ts"; +import { randomInt } from "node:crypto"; + +import { BaseExecutor, type ExecuteInput } from "./base.ts"; import { PROVIDERS } from "../config/constants.ts"; const MODEL_TO_AGENT: Record = { @@ -20,30 +19,39 @@ function generateClientSessionId(): string { const alphabet = "0123456789abcdefghijklmnopqrstuvwxyz"; let out = ""; for (let i = 0; i < 13; i++) { - out += alphabet[Math.floor(Math.random() * alphabet.length)]; + out += alphabet[randomInt(alphabet.length)]; } return out; } export class FreebuffExecutor extends BaseExecutor { constructor() { - super("freebuff", (PROVIDERS as Record).freebuff as string || "freebuff"); + super("freebuff", PROVIDERS.freebuff || { format: "openai" }); } override async execute(input: ExecuteInput) { const { model, body, stream, credentials, signal } = input; const token = credentials?.apiKey || credentials?.accessToken || ""; + const payload = + body && typeof body === "object" && !Array.isArray(body) + ? (body as Record) + : {}; if (!token) { return { response: new Response( - JSON.stringify({ error: { message: "Freebuff Auth Token required", type: "authentication_error" } }), + JSON.stringify({ + error: { message: "Freebuff Auth Token required", type: "authentication_error" }, + }), { status: 401, headers: { "Content-Type": "application/json" } } ), }; } - const requestedModel = typeof model === "string" ? model.replace(/^freebuff\//, "") : (model || "deepseek/deepseek-v4-flash"); + const requestedModel = + typeof model === "string" + ? model.replace(/^freebuff\//, "") + : model || "deepseek/deepseek-v4-flash"; const agentId = MODEL_TO_AGENT[requestedModel] || "base2-free"; const authHeaders = { @@ -73,7 +81,12 @@ export class FreebuffExecutor extends BaseExecutor { const errText = await sessionRes.text(); return { response: new Response( - JSON.stringify({ error: { message: `Freebuff session failed (${sessionRes.status}): ${errText}`, type: "upstream_error" } }), + JSON.stringify({ + error: { + message: `Freebuff session failed (${sessionRes.status}): ${errText}`, + type: "upstream_error", + }, + }), { status: sessionRes.status, headers: { "Content-Type": "application/json" } } ), }; @@ -82,7 +95,9 @@ export class FreebuffExecutor extends BaseExecutor { const msg = e instanceof Error ? e.message : String(e); return { response: new Response( - JSON.stringify({ error: { message: `Freebuff session network error: ${msg}`, type: "upstream_error" } }), + JSON.stringify({ + error: { message: `Freebuff session network error: ${msg}`, type: "upstream_error" }, + }), { status: 502, headers: { "Content-Type": "application/json" } } ), }; @@ -103,12 +118,18 @@ export class FreebuffExecutor extends BaseExecutor { } catch {} // 3. Prepare Chat Payload & Buffy System Prompt - const incomingMessages = Array.isArray(body?.messages) ? [...body.messages] : []; + const incomingMessages: Array> = Array.isArray(payload.messages) + ? payload.messages.filter( + (message): message is Record => + !!message && typeof message === "object" && !Array.isArray(message) + ) + : []; + const firstMessage = incomingMessages[0]; const hasBuffyPrompt = incomingMessages.length > 0 && - incomingMessages[0].role === "system" && - typeof incomingMessages[0].content === "string" && - incomingMessages[0].content.trim().startsWith("You are Buffy"); + firstMessage?.role === "system" && + typeof firstMessage.content === "string" && + firstMessage.content.trim().startsWith("You are Buffy"); if (!hasBuffyPrompt) { incomingMessages.unshift({ @@ -118,8 +139,14 @@ export class FreebuffExecutor extends BaseExecutor { } const clientSessionId = generateClientSessionId(); + const existingMetadata = + payload.codebuff_metadata && + typeof payload.codebuff_metadata === "object" && + !Array.isArray(payload.codebuff_metadata) + ? (payload.codebuff_metadata as Record) + : {}; const upstreamBody = { - ...(body || {}), + ...payload, model: requestedModel, messages: incomingMessages, stream: stream !== false, @@ -128,7 +155,7 @@ export class FreebuffExecutor extends BaseExecutor { cost_mode: "free", client_id: clientSessionId, freebuff_instance_id: instanceId, - ...((body as Record)?.codebuff_metadata as Record || {}), + ...existingMetadata, }, }; diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 6318aaab2e5..59e18719076 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -73,7 +73,7 @@ type GlmEffortTier = { * `thinking.type=enabled` (5.3 no longer accepts thinking disabled). * * https://docs.z.ai/devpack/latest-model - * https://z.ai/blog/glm-5.3 + * https://docs.z.ai/guides/llm/glm-5.3 */ function parseGlmEffortTier(model: string): GlmEffortTier | null { switch (model) { @@ -430,6 +430,7 @@ export class GlmExecutor extends DefaultExecutor { let response: Response; try { + this.assertOutboundUrlAllowed(url); // GHSA-4f49: glm has its own fetch path response = await fetch(url, { method: "POST", headers, diff --git a/open-sse/executors/hailuo-web.ts b/open-sse/executors/hailuo-web.ts index 1d9ffda3588..7d1b839c269 100644 --- a/open-sse/executors/hailuo-web.ts +++ b/open-sse/executors/hailuo-web.ts @@ -1,11 +1,11 @@ /** - * HailuoWebExecutor — Hailuo AI (MiniMax) web chat via www.hailuo.ai. + * HailuoWebExecutor — Hailuo AI (MiniMax) web chat via chat.minimax.io. * * Distinct from the paid API-key `minimax`/`minimax-cn` providers * (open-sse/config/providers/registry/minimax/) — this targets the free - * consumer chat product at hailuo.ai / chat.minimax.io. + * consumer chat product at chat.minimax.io. * - * Endpoint: POST https://www.hailuo.ai/v4/api/chat/msg? + * Endpoint: POST https://chat.minimax.io/v4/api/chat/msg? * Auth: `token` header — value read from the site's `_token` localStorage * entry, plus a per-request `yy` signature header. * Body: multipart/form-data — characterID, msgContent, chatID, searchMode. @@ -33,7 +33,7 @@ import { createHash } from "node:crypto"; import { BaseExecutor, type ExecuteInput } from "./base.ts"; import { makeExecutorErrorResult as makeErrorResult, sanitizeErrorMessage } from "../utils/error.ts"; -const BASE_URL = "https://www.hailuo.ai"; +const BASE_URL = "https://chat.minimax.io"; const API_PATH = "/v4/api/chat/msg"; const USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36"; diff --git a/open-sse/executors/kimi-web.ts b/open-sse/executors/kimi-web.ts index 8f9c13c3328..f2c389822c3 100644 --- a/open-sse/executors/kimi-web.ts +++ b/open-sse/executors/kimi-web.ts @@ -1,10 +1,10 @@ /** - * KimiWebExecutor — Moonshot AI Chat via www.kimi.com (international) + * KimiWebExecutor — Moonshot AI Chat via www.kimi.ai (international) * * Routes requests through Kimi's consumer chat API on the international domain. * Originally this executor targeted `kimi.moonshot.cn` (mainland-CN consumer * chat). That domain now redirects every visitor outside CN to - * `https://www.kimi.com/`, which speaks a completely different API surface: + * `https://www.kimi.ai/`, which speaks a completely different API surface: * * - Endpoint: POST /apiv2/kimi.gateway.chat.v1.ChatService/Chat * - Protocol: Connect-RPC (unary envelope framing — 5-byte header + JSON) @@ -326,7 +326,7 @@ export class KimiWebExecutor extends BaseExecutor { if (!accessToken) { return makeErrorResult( 400, - "Missing Kimi access_token — log in at www.kimi.com and capture access_token from localStorage.", + "Missing Kimi access_token — log in at www.kimi.ai and capture access_token from localStorage.", body, CHAT_URL ); @@ -410,10 +410,7 @@ export class KimiWebExecutor extends BaseExecutor { const refreshToken = credentials?.refreshToken || credentials?.providerSpecificData?.refreshToken; if (refreshToken && typeof refreshToken === "string") { - const refreshRes = await exchangeKimiRefreshToken( - refreshToken, - getKimiWebBaseUrl() - ); + const refreshRes = await exchangeKimiRefreshToken(refreshToken, getKimiWebBaseUrl()); if (refreshRes.success && refreshRes.accessToken) { accessToken = refreshRes.accessToken; const retryHeaders = this.buildKimiHeaders(accessToken); diff --git a/open-sse/executors/muse-spark-web.ts b/open-sse/executors/muse-spark-web.ts index f93191c3124..a8052c4c657 100644 --- a/open-sse/executors/muse-spark-web.ts +++ b/open-sse/executors/muse-spark-web.ts @@ -1070,7 +1070,7 @@ async function wsChat( const fail = (error: string) => finish({ content: "", deltas: [], error }); - timeout = setTimeout(() => fail("Meta AI WebSocket timed out"), 30000); + timeout = setTimeout(() => fail(`Meta AI WS timed out (readyState=${ws.readyState})`), 30000); abortHandler = () => fail("Request aborted"); signal?.addEventListener("abort", abortHandler, { once: true }); diff --git a/open-sse/executors/nlpcloud.ts b/open-sse/executors/nlpcloud.ts index d413b5a683b..e212a38efeb 100644 --- a/open-sse/executors/nlpcloud.ts +++ b/open-sse/executors/nlpcloud.ts @@ -471,6 +471,7 @@ export class NlpCloudExecutor extends BaseExecutor { } try { + this.assertOutboundUrlAllowed(url); // GHSA-4f49: nlpcloud has its own fetch path const response = await fetch(url, { method: "POST", headers, diff --git a/open-sse/executors/opencode.ts b/open-sse/executors/opencode.ts index 51419ead9c5..0829bfa871d 100644 --- a/open-sse/executors/opencode.ts +++ b/open-sse/executors/opencode.ts @@ -1,6 +1,6 @@ import { BaseExecutor, type ExecuteInput, type ProviderCredentials } from "./base.ts"; import { PROVIDERS } from "../config/constants.ts"; -import { getModelTargetFormat } from "../config/providerModels.ts"; +import { getModelTargetFormat, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.ts"; import { injectReasoningContentForThinkingModel, isThinkingMessageModel, @@ -15,6 +15,8 @@ import { markSuccess as markAccountSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, } from "./accountRotation.ts"; import { isNetworkRotationSharedEgressGuardEnabled } from "@/shared/utils/featureFlags"; @@ -125,6 +127,24 @@ export function isPremiumOpencodeModel(model: string, provider: string): boolean return !OPENCODE_FREE_MODELS.has(model); } +/** + * Resolves the registry `targetFormat` for a model, aliasing `provider` first. + * + * `PROVIDER_MODELS` is keyed by the provider's public ALIAS (e.g. `"oc"`), not its + * raw registry id (e.g. `"opencode"`) — mirrors `resolveChatCoreTargetFormat()` + * (`handlers/chatCore/targetFormat.ts`), which already aliases before calling + * `getModelTargetFormat()`. Calling it with the raw id here made every entry miss + * silently (fell through to `"openai"`), while chatCore's own request-body + * translation (correctly aliased) still switched to the Responses API shape for + * `targetFormat:"openai-responses"` models — sending a Responses-shaped body to + * the `/chat/completions` URL this executor's own `buildUrl()` kept selecting. + * Exported for testability. + */ +export function resolveOpencodeTargetFormat(provider: string, model: string): string { + const alias = PROVIDER_ID_TO_ALIAS[provider] || provider; + return getModelTargetFormat(alias, model) || "openai"; +} + export class OpencodeExecutor extends BaseExecutor { /** Delegates to `isPremiumOpencodeModel`. Exported for testability. */ static isPremiumModel(model: string, provider: string): boolean { @@ -193,8 +213,11 @@ export class OpencodeExecutor extends BaseExecutor { return pickRotatableAccount(this.accounts, this); } - private markCooldown(account: OpencodeAccountState): void { - markAccountCooldown(account); + private markCooldown( + account: OpencodeAccountState, + kind: "transient" | "terminal" = "transient" + ): void { + markAccountCooldown(account, kind); } private markSuccess(account: OpencodeAccountState): void { @@ -202,7 +225,7 @@ export class OpencodeExecutor extends BaseExecutor { } async execute(input: ExecuteInput) { - this._requestFormat = getModelTargetFormat(this.provider, input.model) || "openai"; + this._requestFormat = resolveOpencodeTargetFormat(this.provider, input.model); // #8681: Gate premium opencode models behind a usable API key. // When the connection is keyless (no apiKey, no accessToken) and the model @@ -232,14 +255,41 @@ export class OpencodeExecutor extends BaseExecutor { try { this.syncAccountsFromCredentials(input.credentials); + const { log } = input; const hasProxies = this.accounts.some((a) => a.proxy !== null); - // Fast path: no multi-account proxy wiring configured → original behavior. + // Fast path: no multi-account proxy wiring configured → original behavior, + // plus exactly ONE bounded retry when the upstream answers a 400 empty + // rejection (same predicate and logging as the rotation loop). Everything + // else passes untouched: this path deliberately preserves BaseExecutor's + // intra-URL 429 retries (no skipUpstreamRetry here). if (this.accounts.length === 1 && !hasProxies) { - return await super.execute(input); + const single = (await super.execute(input)) as HttpExecuteResult; + if (single.response.status === 400) { + let bodyText: string | null = null; + try { + bodyText = await single.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on direct account"); + } + if (bodyText !== null) { + if (isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on direct account (${chatcmplId}), retrying once…` + ); + return await super.execute(input); + } + log?.debug?.( + "OPENCODE", + "400 without error field, signature not matched on direct account — observing" + ); + } + } + return single; } - const { log } = input; // This loop only ever dispatches through super.execute() (the HTTP request // path), which always resolves the object-shaped arm of ExecutorExecuteResult // — the bare-Response arm belongs to web/scraping executors only (base.ts:290). @@ -256,8 +306,13 @@ export class OpencodeExecutor extends BaseExecutor { // network call, but proxied accounts (independent egress) are still // tried normally. let sharedEgressDown = false; + // Bounded extra attempts for empty upstream rejections: +1 for a single + // account (retry the same one), none for a multi-account fleet (rotation + // through the accounts is the retry). Avoids an unbounded loop on a + // persistently malformed upstream. + const emptyRejectionBudget = this.accounts.length === 1 ? 1 : 0; - for (let attempt = 0; attempt < this.accounts.length; attempt++) { + for (let attempt = 0; attempt < this.accounts.length + emptyRejectionBudget; attempt++) { const account = this.pickAccount(); const masked = maskAccountId(account.fingerprint); @@ -333,6 +388,34 @@ export class OpencodeExecutor extends BaseExecutor { continue; } + // Empty upstream rejection (malformed 400: no error field, no real + // content, finish_reason null — see isEmptyUpstreamRejection). Rotate/ + // retry instead of propagating it as a fatal success: the observed + // envelope was marking subagent sessions as failed. Read the body ONLY + // for a 400 (never a 200/streaming — that would buffer the good path); + // classify, log, and continue. Neitheries markCooldown nor markSuccess: + // the failure is upstream's, not this account's. + if (status === 400) { + let bodyText: string | null = null; + try { + bodyText = await result.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on empty rejection check"); + } + if (bodyText !== null && isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on account ${masked} (${chatcmplId}), rotating to next…` + ); + continue; + } + // A 400 carrying a real error (or non-empty content): propagate + // immediately, untouched — same as before this change. + this.markSuccess(account); + return result; + } + this.markSuccess(account); return result; } diff --git a/open-sse/executors/zcode.ts b/open-sse/executors/zcode.ts index 0841b4daa8e..8f0a98ac142 100644 --- a/open-sse/executors/zcode.ts +++ b/open-sse/executors/zcode.ts @@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto"; import { existsSync } from "node:fs"; import { homedir } from "node:os"; import { join, resolve } from "node:path"; -import { GLM_SHARED_MODELS } from "../config/glmProvider.ts"; +import { ZCODE_MODELS } from "../config/providers/registry/zcode/index.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorExecuteResult, type ProviderCredentials } from "./base.ts"; import { ZcodeAppServerClient, type ZcodeClientLike } from "./zcodeProtocol.ts"; import { buildErrorBody, errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; @@ -12,8 +12,8 @@ const DEFAULT_PROVIDER_ID = "builtin:zai-coding-plan"; const DEFAULT_TURN_TIMEOUT_MS = 120_000; const DEFAULT_POLL_INTERVAL_MS = 250; const TERMINAL_STATUSES = new Set(["completed", "idle", "paused", "error"]); -const ZCODE_MODEL_ALLOWLIST = new Set(GLM_SHARED_MODELS.map((model) => model.id)); -const DEFAULT_ZCODE_MODEL = GLM_SHARED_MODELS[0]?.id || "glm-5.2"; +const ZCODE_MODEL_ALLOWLIST = new Set(ZCODE_MODELS.map((model) => model.id)); +const DEFAULT_ZCODE_MODEL = ZCODE_MODELS[0]?.id || "glm-5.2"; type JsonRecord = Record; type OpenAIMsg = { role?: string; content?: unknown }; diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 9f530d46056..8f11373b1fb 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -367,7 +367,10 @@ import { resolveReportedServiceTier as resolveReportedServiceTierFor, type EffectiveServiceTier, } from "./chatCore/serviceTier.ts"; -import { cacheReasoningFromAssistantMessage } from "../services/reasoningCache.ts"; +import { + cacheReasoningFromAssistantMessage, + requiresReasoningReplay, +} from "../services/reasoningCache.ts"; import { sanitizeOpenAITool } from "../services/toolSchemaSanitizer.ts"; import { isCompactResponsesEndpoint } from "../executors/codex.ts"; import { persistCodexChildQuotaResponse } from "../services/codexAccount/index.ts"; @@ -434,6 +437,7 @@ import { import { generateRequestId } from "@/shared/utils/requestId"; import { isLocalStreamLifecycleError } from "@/shared/utils/circuitBreaker"; import { shouldIsolateProbeFailures } from "@/shared/utils/probeOrigin"; +import { writeTerminalStatus } from "@/shared/utils/terminalStatus"; import { extractFacts } from "@/lib/memory/extraction"; import { handleToolCallExecution } from "@/lib/skills/interception"; import { MEMORY_BUILTIN_TOOL_NAMES } from "@/lib/skills/memoryBuiltins"; @@ -2641,12 +2645,16 @@ export async function handleChatCore({ // no-op. #7694: `modelInfo.resolvedThinkingEffort` — set when the request's model // id carried a `/-{effort}` synced-model alias suffix // (`src/sse/services/model.ts`) — takes priority over the static per-model default. - // See open-sse/services/defaultReasoningEffort.ts. + // The synced catalog's vendor-declared `defaultThinkingEffort` (OpenRouter + // `reasoning.default_effort`, captured by `detectDefaultThinkingEffort`) is the + // lowest-priority default: it only fires when neither the suffix alias nor a + // static operator default exists. See open-sse/services/defaultReasoningEffort.ts. if (targetFormat === FORMATS.OPENAI) { translatedBody = applyDefaultReasoningEffort( translatedBody, finalModelToUpstream, - (modelInfo as { resolvedThinkingEffort?: string })?.resolvedThinkingEffort + (modelInfo as { resolvedThinkingEffort?: string })?.resolvedThinkingEffort, + (modelInfo as { defaultThinkingEffort?: string })?.defaultThinkingEffort ); } } @@ -3066,7 +3074,9 @@ export async function handleChatCore({ executor, provider, model: modelToCall, - connectionTimeoutMs: resolveConnectionTimeoutMs(execCreds?.providerSpecificData), + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), signal: streamController.signal, log, execute: (signal) => @@ -3370,7 +3380,9 @@ export async function handleChatCore({ executor, provider, model: modelToCall, - connectionTimeoutMs: resolveConnectionTimeoutMs(execCreds?.providerSpecificData), + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), signal: streamController.signal, log, execute: (signal) => @@ -4113,29 +4125,28 @@ export async function handleChatCore({ if (errorConnectionId && errorType) { try { if (errorType === PROVIDER_ERROR_TYPES.FORBIDDEN) { - // T-PROBE: a probe-origin failure (model test-all) must never - // remove the connection from the pool — record but stay active. - if (await shouldIsolateProbeFailures()) { - await updateProviderConnection(errorConnectionId, { - lastErrorType: errorType, - lastError: message, - errorCode: statusCode, - lastErrorAt: new Date().toISOString(), - }); - console.warn( - `[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active` - ); - } else { - await updateProviderConnection(errorConnectionId, { - isActive: false, - testStatus: "banned", - lastErrorType: errorType, - lastError: message, - errorCode: statusCode, - }); - console.warn( - `[provider] Node ${errorConnectionId} banned (${statusCode}) — disabling permanently` + { + const probeIsolated = await shouldIsolateProbeFailures(); + await writeTerminalStatus( + errorConnectionId, + { + testStatus: "banned", + isActive: false, + lastError: message, + lastErrorType: errorType, + errorCode: String(statusCode), + }, + probeIsolated ? "probe" : "production" ); + if (probeIsolated) { + console.warn( + `[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active` + ); + } else { + console.warn( + `[provider] Node ${errorConnectionId} banned (${statusCode}) — disabling permanently` + ); + } } } else if (errorType === PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED) { // T-PROBE: probe-origin failures (test-all) never deactivate — @@ -4158,44 +4169,47 @@ export async function handleChatCore({ console.warn( `[provider] Node ${errorConnectionId} account deactivated (${statusCode}) — has extra keys, keeping connection active` ); - } else if (await shouldIsolateProbeFailures()) { - await updateProviderConnection(errorConnectionId, { - lastErrorType: errorType, - lastError: message, - errorCode: statusCode, - lastErrorAt: new Date().toISOString(), - }); - console.warn( - `[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active` - ); } else { - await updateProviderConnection(errorConnectionId, { - isActive: false, - testStatus: "deactivated", - lastErrorType: errorType, - lastError: message, - errorCode: statusCode, - }); - console.warn( - `[provider] Node ${errorConnectionId} account deactivated (${statusCode}) — disabling permanently` + const probeIsolated2 = await shouldIsolateProbeFailures(); + await writeTerminalStatus( + errorConnectionId, + { + testStatus: "deactivated", + isActive: false, + lastError: message, + lastErrorType: errorType, + errorCode: String(statusCode), + }, + probeIsolated2 ? "probe" : "production" ); + if (probeIsolated2) { + console.warn( + `[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active` + ); + } else { + console.warn( + `[provider] Node ${errorConnectionId} account deactivated (${statusCode}) — disabling permanently` + ); + } } } else if (errorType === PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED) { - // T-PROBE: probe-origin failures never write quota state — - // `testStatus: "credits_exhausted"` is terminal and removes the - // connection from the pool; semaphore locks and per-model quota - // lockouts are routing mutations too. Record only (#9817). - if (await shouldIsolateProbeFailures()) { - await updateProviderConnection(errorConnectionId, { - lastErrorType: errorType, - lastError: message, - errorCode: statusCode, - lastErrorAt: new Date().toISOString(), - }); - console.warn( - `[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active` - ); - } else { + { + const probeIsolated3 = await shouldIsolateProbeFailures(); + if (probeIsolated3) { + await writeTerminalStatus( + errorConnectionId, + { + testStatus: "credits_exhausted", + lastError: message, + lastErrorType: errorType, + errorCode: String(statusCode), + }, + "probe" + ); + console.warn( + `[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active` + ); + } else { // Kimi's 403 says "billing cycle" for both an exhausted subscription and a // temporary request window. Read its official usage endpoint before making // the connection terminal: a non-zero Weekly quota plus an empty Ratelimit @@ -4257,14 +4271,19 @@ export async function handleChatCore({ `[provider] Node ${errorConnectionId} ${quotaScope}-only quota exhausted (${statusCode}) for ${model} - ${Math.ceil(quotaCooldownMs / 1000)}s (cooldown_scope=${quotaScope}, ttl_source=${retryAfterMs ? "upstream" : "inferred"}, connection stays active)` ); } else { - await updateProviderConnection(errorConnectionId, { - testStatus: "credits_exhausted", - lastErrorType: errorType, - lastError: message, - errorCode: statusCode, - }); + await writeTerminalStatus( + errorConnectionId, + { + testStatus: "credits_exhausted", + lastError: message, + lastErrorType: errorType, + errorCode: String(statusCode), + }, + "production" + ); console.warn(`[provider] Node ${errorConnectionId} exhausted quota (${statusCode})`); } + } // close probeIsolated3 else } } else if (errorType === PROVIDER_ERROR_TYPES.UNAUTHORIZED) { // Normal 401 (token/session auth issue): keep account active for refresh/re-auth. @@ -4918,10 +4937,12 @@ export async function handleChatCore({ const msg = firstChoice?.message; const historyMessages = (translatedBody as { messages?: unknown[] } | null | undefined) ?.messages; - cacheReasoningFromAssistantMessage(msg, provider, model, { - scope: reasoningCacheScope, - historyMessages: Array.isArray(historyMessages) ? historyMessages : [], - }); + if (requiresReasoningReplay({ provider, model })) { + cacheReasoningFromAssistantMessage(msg, provider, model, { + scope: reasoningCacheScope, + historyMessages: Array.isArray(historyMessages) ? historyMessages : [], + }); + } } catch { // Cache capture is non-critical — never block the response } @@ -5440,10 +5461,12 @@ export async function handleChatCore({ const msg = choices?.[0]?.message; const historyMessages = (translatedBody as { messages?: unknown[] } | null | undefined) ?.messages; - cacheReasoningFromAssistantMessage(msg, provider, model, { - scope: reasoningCacheScope, - historyMessages: Array.isArray(historyMessages) ? historyMessages : [], - }); + if (requiresReasoningReplay({ provider, model })) { + cacheReasoningFromAssistantMessage(msg, provider, model, { + scope: reasoningCacheScope, + historyMessages: Array.isArray(historyMessages) ? historyMessages : [], + }); + } } catch { // Cache capture is non-critical — never block the stream } diff --git a/open-sse/handlers/rerank.ts b/open-sse/handlers/rerank.ts index 747e1ce5066..452e6f3500e 100644 --- a/open-sse/handlers/rerank.ts +++ b/open-sse/handlers/rerank.ts @@ -199,6 +199,8 @@ export async function handleRerank({ return_documents, credentials, connectionId = null, + apiKeyId = null, + apiKeyName = null, }) { const startTime = Date.now(); if (!model) return errorResponse(400, "model is required"); @@ -267,10 +269,23 @@ export async function handleRerank({ if (!res.ok) { const errData = await res.json().catch(() => ({})); - return errorResponse( - res.status, - errData.message || errData.error?.message || `Provider returned HTTP ${res.status}` - ); + const errorMessage = + errData.message || errData.error?.message || `Provider returned HTTP ${res.status}`; + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: res.status, + model: `${providerId}/${modelId}`, + provider: providerId, + connectionId: connectionId || undefined, + duration: Date.now() - startTime, + requestBody, + responseBody: errData, + error: errorMessage, + apiKeyId: apiKeyId || undefined, + apiKeyName: apiKeyName || undefined, + }).catch(() => {}); + return errorResponse(res.status, errorMessage); } const data = await res.json(); @@ -289,10 +304,13 @@ export async function handleRerank({ status: 200, model: `${providerId}/${modelId}`, provider: providerId, + connectionId: connectionId || undefined, duration: Date.now() - startTime, tokens: { prompt_tokens: 0, completion_tokens: 0 }, - responseBody: { results_count: Array.isArray(result?.results) ? result.results.length : 0 }, - connectionId, + requestBody, + responseBody: result, + apiKeyId: apiKeyId || undefined, + apiKeyName: apiKeyName || undefined, }).catch(() => {}); const headers = new Headers({ ...CORS_HEADERS, "Content-Type": "application/json" }); diff --git a/open-sse/handlers/responseTranslator.ts b/open-sse/handlers/responseTranslator.ts index 01bdd4c14a7..43e03d919d6 100644 --- a/open-sse/handlers/responseTranslator.ts +++ b/open-sse/handlers/responseTranslator.ts @@ -10,6 +10,7 @@ import { caseInsensitiveToolNameLookup, restoreOpenAIToolNames, } from "../translator/helpers/toolCallHelper.ts"; +import { restoreClaudeToolName } from "../services/claudeCodeToolRemapper.ts"; import { extractReplayableResponsesReasoningText } from "../services/reasoningInputPolicy.ts"; import { sanitizeToolId } from "../translator/helpers/schemaCoercion.ts"; @@ -631,7 +632,7 @@ export function translateNonStreamingResponse( // Phase 3: Translate from OpenAI back to Client Source format if (sourceFormat === FORMATS.CLAUDE && sourceFormat !== targetFormat) { - return convertOpenAINonStreamingToClaude(toRecord(intermediateOpenAI)); + return convertOpenAINonStreamingToClaude(toRecord(intermediateOpenAI), toolNameMap ?? null); } // Gemini-family clients (Gemini, Antigravity): the streaming SSE path already @@ -667,8 +668,18 @@ function resolveReasoningText(messageObj: JsonRecord): string { /** * Helper to convert an OpenAI chat.completion JSON object to Claude format for non-streaming. + * + * `toolNameMap` carries request-side aliases; when it does not resolve a name, + * `restoreClaudeToolName` upgrades known Claude Code tools to their canonical + * PascalCase ("bash" → "Bash", "croncreate" → "CronCreate"). Without this, a + * non-streaming upstream JSON body (or a stream:true request the upstream + * answered with application/json) reaches Claude Code with lowercase tool_use + * names the CLI rejects as "No such tool available". */ -function convertOpenAINonStreamingToClaude(openaiResponse: JsonRecord): JsonRecord { +function convertOpenAINonStreamingToClaude( + openaiResponse: JsonRecord, + toolNameMap?: Map | null +): JsonRecord { const choices = openaiResponse.choices as unknown[] | undefined; const isChoicesArray = Array.isArray(choices); if (!isChoicesArray && openaiResponse.object !== "chat.completion") { @@ -717,7 +728,7 @@ function convertOpenAINonStreamingToClaude(openaiResponse: JsonRecord): JsonReco content.push({ type: "tool_use", id: sanitizeToolId(rawId), - name: toString(fn.name), + name: restoreClaudeToolName(toString(fn.name), toolNameMap ?? null), input: typeof fn.arguments === "string" ? JSON.parse(fn.arguments || "{}") : fn.arguments || {}, }); diff --git a/open-sse/handlers/search.ts b/open-sse/handlers/search.ts index ca6221538a2..288882ae14a 100644 --- a/open-sse/handlers/search.ts +++ b/open-sse/handlers/search.ts @@ -7,22 +7,27 @@ import { randomUUID } from "crypto"; * serper-search, brave-search, perplexity-search, exa-search, tavily-search, * firecrawl, google-pse-search, linkup-search, searchapi-search, * youcom-search, searxng-search, ollama-search, zai-search, jina-search, - * duckduckgo-free + * duckduckgo-free, x-search (Grok / SuperGrok X Search — explicit or search_type "x") * * Request format: * { * "query": "search query", * "provider": "serper-search" | "brave-search" | ... // optional, auto-selects cheapest * "max_results": 5, - * "search_type": "web" | "news" + * "search_type": "web" | "news" | "x" * } */ -import { getSearchProvider, type SearchProviderConfig } from "../config/searchRegistry.ts"; +import { + getSearchProvider, + isUnconfiguredLoopbackSearchProvider, + type SearchProviderConfig, +} from "../config/searchRegistry.ts"; import { buildPerplexityRequest, parsePerplexitySearchOptions } from "./search/perplexitySearch.ts"; import * as fcSearch from "./search/firecrawlSearch.ts"; import { type FirecrawlSearchEnvelope } from "./search/firecrawlSearch.ts"; import { buildJinaSearchRequest, extractJinaSearchItems } from "./search/jinaSearch.ts"; +import * as xSearch from "./search/xSearch.ts"; import { freeWebSearch } from "../services/freeWebSearch.ts"; import { saveCallLog } from "@/lib/usageDb"; import { safeOutboundFetch } from "@/shared/network/safeOutboundFetch"; @@ -629,6 +634,7 @@ const requestBuilders: Record = { "searxng-search": buildSearxngRequest, "ollama-search": buildOllamaRequest, "jina-search": buildJinaSearchRequest, + "x-search": xSearch.buildXSearchRequest, }; function buildRequest( @@ -1203,6 +1209,7 @@ const responseNormalizers: Record = { "searxng-search": normalizeSearxngResponse, "ollama-search": normalizeOllamaResponse, "jina-search": normalizeJinaSearchResponse, + "x-search": normalizeXSearchResponse, }; function normalizeResponse( @@ -1213,10 +1220,33 @@ function normalizeResponse( ): { results: SearchResult[]; totalResults: number | null } { const normalizer = responseNormalizers[providerId]; if (normalizer) return normalizer(data, query, searchType); - return { results: [], totalResults: null }; } +function normalizeXSearchResponse( + data: unknown, + query: string, + _searchType: string +): { results: SearchResult[]; totalResults: number | null } { + const now = new Date().toISOString(); + const hits = xSearch.extractXSearchHits(data, query, 20); + const results = hits.map((hit, idx) => + makeResult( + "x-search", + { + title: hit.title, + url: hit.url, + snippet: hit.snippet, + author: hit.author, + source_type: "x", + }, + idx, + now + ) + ); + return { results, totalResults: results.length }; +} + function normalizeJinaSearchResponse( data: unknown, _query: string, @@ -1309,7 +1339,60 @@ export async function handleSearch(options: SearchHandlerOptions): Promise; + providerSpecificData?: Record; +} + +export type XSearchHit = { + title: string; + url: string; + snippet: string; + author?: string; +}; + +const X_POST_URL_RE = /^https?:\/\/(?:www\.)?(?:x|twitter)\.com\/([^/?#]+)\/status\/(\d+)/i; +const X_PROFILE_URL_RE = /^https?:\/\/(?:www\.)?(?:x|twitter)\.com\/([^/?#]+)\/?$/i; + +function asRecord(value: unknown): Record | null { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : null; +} + +function timeRangeToFromDate(timeRange?: string): string | undefined { + if (!timeRange || timeRange === "any" || timeRange === "hour") return undefined; + const now = Date.now(); + const day = 24 * 60 * 60 * 1000; + const deltas: Record = { + day: day, + week: 7 * day, + month: 30 * day, + year: 365 * day, + }; + const delta = deltas[timeRange]; + if (!delta) return undefined; + return new Date(now - delta).toISOString().slice(0, 10); +} + +function handlesFromDomainFilter(domainFilter?: string[]): string[] | undefined { + if (!domainFilter?.length) return undefined; + const handles = domainFilter + .filter((d) => !d.startsWith("-")) + .map((d) => d.replace(/^@/, "").replace(/^(?:www\.)?(?:x|twitter)\.com\//i, "").split("/")[0]) + .filter((h) => /^[A-Za-z0-9_]{1,15}$/.test(h)) + .slice(0, 20); + return handles.length ? handles : undefined; +} + +export function titleFromXUrl(url: string): string { + const post = url.match(X_POST_URL_RE); + if (post) return `@${post[1]}`; + const profile = url.match(X_PROFILE_URL_RE); + if (profile && !["i", "intent", "share", "search"].includes(profile[1].toLowerCase())) { + return `@${profile[1]}`; + } + return "X post"; +} + +function addUrl(urls: string[], seen: Set, raw: unknown): void { + if (typeof raw !== "string") return; + const url = raw.trim(); + if (!url.startsWith("http")) return; + if (seen.has(url)) return; + seen.add(url); + urls.push(url); +} + +function walkForUrls(value: unknown, urls: string[], seen: Set, depth = 0): void { + if (depth > 8 || value == null) return; + if (typeof value === "string") { + if (/^https?:\/\//.test(value) && /(?:x|twitter)\.com\//i.test(value)) { + addUrl(urls, seen, value); + } + return; + } + if (Array.isArray(value)) { + for (const item of value) walkForUrls(item, urls, seen, depth + 1); + return; + } + const rec = asRecord(value); + if (!rec) return; + for (const key of ["url", "uri", "href", "source"]) { + addUrl(urls, seen, rec[key]); + } + for (const nested of Object.values(rec)) walkForUrls(nested, urls, seen, depth + 1); +} + +export function extractXSearchHits( + data: unknown, + query: string, + maxResults: number +): XSearchHit[] { + const rec = asRecord(data) ?? {}; + const urls: string[] = []; + const seen = new Set(); + + if (Array.isArray(rec.citations)) { + for (const c of rec.citations) addUrl(urls, seen, c); + } + + walkForUrls(rec.output, urls, seen); + walkForUrls(rec.output_text, urls, seen); + + let snippet = ""; + if (typeof rec.output_text === "string") snippet = rec.output_text.trim(); + if (!snippet && Array.isArray(rec.output)) { + for (const item of rec.output) { + const row = asRecord(item); + if (!row) continue; + if (typeof row.text === "string" && row.text.trim()) { + snippet = row.text.trim(); + break; + } + if (Array.isArray(row.content)) { + for (const part of row.content) { + const p = asRecord(part); + if (p && typeof p.text === "string" && p.text.trim()) { + snippet = p.text.trim(); + break; + } + } + } + if (snippet) break; + } + } + if (!snippet) snippet = query; + + const xUrls = urls.filter((u) => /(?:x|twitter)\.com\//i.test(u)); + const chosen = (xUrls.length ? xUrls : urls).slice(0, maxResults); + + return chosen.map((url) => ({ + title: titleFromXUrl(url), + url, + snippet: snippet.slice(0, 500), + author: titleFromXUrl(url).startsWith("@") ? titleFromXUrl(url).slice(1) : undefined, + })); +} + +export function buildXSearchRequest( + config: SearchProviderConfig, + params: XSearchParams +): { url: string; init: RequestInit } { + const model = + (typeof params.providerSpecificData?.model === "string" && + params.providerSpecificData.model.trim()) || + (typeof params.providerOptions?.model === "string" && params.providerOptions.model.trim()) || + DEFAULT_X_SEARCH_MODEL; + + const tool: Record = { type: "x_search" }; + const fromDate = timeRangeToFromDate(params.timeRange); + if (fromDate) tool.from_date = fromDate; + const handles = handlesFromDomainFilter(params.domainFilter); + if (handles) tool.allowed_x_handles = handles; + + const url = (config.baseUrl || X_SEARCH_RESPONSES_URL).replace(/\/+$/, ""); + return { + url, + init: { + method: "POST", + headers: { + "Content-Type": "application/json", + Accept: "application/json", + ...(params.token ? { Authorization: `Bearer ${params.token}` } : {}), + }, + body: JSON.stringify({ + model, + stream: false, + input: params.query, + tools: [tool], + }), + }, + }; +} diff --git a/open-sse/mcp-server/__tests__/essentialTools.test.ts b/open-sse/mcp-server/__tests__/essentialTools.test.ts index 4948c47ffbb..9efd7ce110f 100644 --- a/open-sse/mcp-server/__tests__/essentialTools.test.ts +++ b/open-sse/mcp-server/__tests__/essentialTools.test.ts @@ -22,10 +22,10 @@ describe("MCP Essential Tools", () => { }); describe("Tool schema validation", () => { - it("should have exactly 13 essential tools (including Radar catalog)", () => { - // 12 -> 13: F3 shipped omniroute_radar_catalog as a phase-1 read-only tool. + it("should have exactly 14 essential tools (including Radar catalog + x_search)", () => { + // 13 -> 14: #10985 shipped omniroute_x_search as a phase-1 tool. const schemas = MCP_ESSENTIAL_TOOLS; - expect(schemas).toHaveLength(13); + expect(schemas).toHaveLength(14); }); it("all tools should have omniroute_ prefix", () => { @@ -297,6 +297,69 @@ describe("omniroute_web_search handler (via MCP dispatch)", () => { }); }); +describe("omniroute_x_search handler (via MCP dispatch)", () => { + let client: Client; + + beforeEach(async () => { + mockFetch.mockReset(); + + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair(); + const server = createMcpServer(); + await server.connect(serverTransport); + client = new Client({ name: "test-client", version: "1.0.0" }); + await client.connect(clientTransport); + }); + + afterEach(async () => { + await client.close(); + }); + + it("should appear in tools/list after registration", async () => { + const { tools } = await client.listTools(); + const xSearch = tools.find((t) => t.name === "omniroute_x_search"); + expect(xSearch).toBeDefined(); + expect(xSearch?.description).toMatch(/X \(Twitter\)/i); + }); + + it("should POST to /v1/search with provider x-search and search_type x", async () => { + mockFetch.mockResolvedValueOnce({ + ok: true, + json: async () => ({ + id: "xs1", + provider: "x-search", + query: "SuperGrok", + results: [ + { + title: "@xai", + url: "https://x.com/xai/status/1", + snippet: "Cited SuperGrok discussion.", + position: 1, + }, + ], + cached: false, + usage: { queries_used: 1, search_cost_usd: 0 }, + }), + }); + + const result = await client.callTool({ + name: "omniroute_x_search", + arguments: { query: "SuperGrok", max_results: 5 }, + }); + + expect(result.isError).toBeFalsy(); + expect(mockFetch).toHaveBeenCalledWith( + expect.stringContaining("/v1/search"), + expect.objectContaining({ method: "POST" }) + ); + const [, options] = mockFetch.mock.calls[0]; + const body = JSON.parse(options.body as string); + expect(body.query).toBe("SuperGrok"); + expect(body.max_results).toBe(5); + expect(body.search_type).toBe("x"); + expect(body.provider).toBe("x-search"); + }); +}); + // ── omniroute_get_health: handler dispatch tests ────────────────────────────── // These tests use InMemoryTransport + Client to exercise the actual registered // handler (not mockFetch directly), so they catch the real bug the original diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index 8b7f3cee324..50c16fa9665 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -106,6 +106,29 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("declares exact GLM reasoning-effort tiers across every shared GLM provider", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt"]) { + for (const model of getModelsByProviderId(provider)) { + expect(model.supportedThinkingEfforts, `${provider}/${model.id} effort tiers`).toEqual( + routedTiers.get(model.id) ?? [] + ); + } + } + + for (const model of getModelsByProviderId("zcode")) { + expect(model.supportedThinkingEfforts, `zcode/${model.id} effort tiers`).toEqual([]); + } + }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); diff --git a/open-sse/mcp-server/__tests__/mcp-runtime-blocked-provider-schema.test.ts b/open-sse/mcp-server/__tests__/mcp-runtime-blocked-provider-schema.test.ts new file mode 100644 index 00000000000..9f55c6a5208 --- /dev/null +++ b/open-sse/mcp-server/__tests__/mcp-runtime-blocked-provider-schema.test.ts @@ -0,0 +1,58 @@ +import { describe, it, expect } from "vitest"; +import { createMcpServer } from "../server"; +import { buildWebSearchInputSchema } from "../schemas/tools"; +import { getActiveSearchProviders } from "../schemas/providerEnums"; + +interface ToolWithSchema { + inputSchema: { + safeParse: (arg: unknown) => { success: boolean }; + }; +} + +describe("MCP Dynamic Runtime Schema Plumbing", () => { + it("getActiveSearchProviders excludes blocked providers dynamically by id or alias", () => { + const allProviders = getActiveSearchProviders([]); + expect(allProviders).toContain("serper-search"); + expect(allProviders).toContain("brave-search"); + + const filteredProviders = getActiveSearchProviders(["serper", "brave"]); + expect(filteredProviders).not.toContain("serper-search"); + expect(filteredProviders).not.toContain("brave-search"); + expect(filteredProviders.length).toBeGreaterThan(0); + }); + + it("buildWebSearchInputSchema excludes blocked providers from Zod enum", () => { + const fullSchema = buildWebSearchInputSchema([]); + const fullParsed = fullSchema.safeParse({ query: "test", provider: "serper-search" }); + expect(fullParsed.success).toBe(true); + + const blockedSchema = buildWebSearchInputSchema(["serper"]); + const blockedParsed = blockedSchema.safeParse({ query: "test", provider: "serper-search" }); + expect(blockedParsed.success).toBe(false); + }); + + it("createMcpServer with blockedProviders option registers dynamic tool schema", async () => { + const server = createMcpServer({ blockedProviders: ["serper", "brave"] }); + expect(server).toBeTruthy(); + + const registeredTools = ( + server as unknown as { _registeredTools: Record } + )._registeredTools; + expect(registeredTools).toBeTruthy(); + + const webSearchTool = registeredTools["omniroute_web_search"]; + expect(webSearchTool).toBeTruthy(); + + const parsedWithUnblocked = webSearchTool.inputSchema.safeParse({ + query: "test", + provider: "perplexity-search", + }); + expect(parsedWithUnblocked.success).toBe(true); + + const parsedWithBlocked = webSearchTool.inputSchema.safeParse({ + query: "test", + provider: "serper-search", + }); + expect(parsedWithBlocked.success).toBe(false); + }); +}); diff --git a/open-sse/mcp-server/schemas/providerEnums.ts b/open-sse/mcp-server/schemas/providerEnums.ts index adb8da067e7..61e4648c082 100644 --- a/open-sse/mcp-server/schemas/providerEnums.ts +++ b/open-sse/mcp-server/schemas/providerEnums.ts @@ -1,12 +1,16 @@ import { SEARCH_PROVIDERS } from "../../config/searchRegistry"; +import { isProviderBlockedByIdOrAlias } from "../../../src/shared/utils/noAuthProviders"; /** * Dynamically generates a tuple of active search provider IDs for Zod enums. - * Filters out any providers marked as disabled in the registry. + * Filters out any providers marked as disabled or blocked in the security policy. */ -export function getActiveSearchProviders(): [string, ...string[]] { +export function getActiveSearchProviders(blockedProviders: string[] = []): [string, ...string[]] { const activeProviders = Object.values(SEARCH_PROVIDERS) - .filter((provider) => !provider.disabled) + .filter( + (provider) => + !provider.disabled && !isProviderBlockedByIdOrAlias(provider.id, blockedProviders) + ) .map((provider) => provider.id); if (activeProviders.length === 0) { diff --git a/open-sse/mcp-server/schemas/tools.ts b/open-sse/mcp-server/schemas/tools.ts index 516b866ec36..d8d83c17ea5 100644 --- a/open-sse/mcp-server/schemas/tools.ts +++ b/open-sse/mcp-server/schemas/tools.ts @@ -462,25 +462,29 @@ export const listModelsCatalogTool: McpToolDefinition< }; // --- Tool 10: omniroute_web_search --- -export const webSearchInput = z.object({ - query: z - .string() - .min(1, "Query is required") - .max(500, "Query must be 500 characters or fewer") - .describe("The search query string"), - max_results: z - .number() - .int() - .min(1) - .max(20) - .default(5) - .describe("Maximum number of search results to return"), - search_type: z.enum(["web", "news"]).default("web").describe("Type of search to perform"), - provider: z - .enum(getActiveSearchProviders()) - .optional() - .describe("Specific search provider to use"), -}); +export function buildWebSearchInputSchema(blockedProviders: string[] = []) { + return z.object({ + query: z + .string() + .min(1, "Query is required") + .max(500, "Query must be 500 characters or fewer") + .describe("The search query string"), + max_results: z + .number() + .int() + .min(1) + .max(20) + .default(5) + .describe("Maximum number of search results to return"), + search_type: z.enum(["web", "news"]).default("web").describe("Type of search to perform"), + provider: z + .enum(getActiveSearchProviders(blockedProviders)) + .optional() + .describe("Specific search provider to use"), + }); +} + +export const webSearchInput = buildWebSearchInputSchema(); export const webSearchOutput = z.object({ id: z.string(), @@ -505,7 +509,7 @@ export const webSearchOutput = z.object({ export const webSearchTool: McpToolDefinition = { name: "omniroute_web_search", description: - "Performs a web search using OmniRoute's search gateway. Supports multiple providers (Serper, Brave, Perplexity, Exa, Tavily, Google PSE, Linkup, SearchAPI, SearXNG) with automatic failover. Returns search results with titles, URLs, snippets, and position data.", + "Performs a web search using OmniRoute's search gateway. Supports multiple providers (Serper, Brave, Perplexity, Exa, Tavily, Google PSE, Linkup, SearchAPI, SearXNG) with automatic failover. Returns search results with titles, URLs, snippets, and position data. Not X/Twitter — use omniroute_x_search for that.", inputSchema: webSearchInput, outputSchema: webSearchOutput, scopes: ["execute:search"], @@ -514,6 +518,33 @@ export const webSearchTool: McpToolDefinition = { + name: "omniroute_x_search", + description: + "Search X (Twitter) through OmniRoute using SuperGrok / xAI server-side x_search. Requires a connected xai-oauth (SuperGrok) or xAI API key. This is Grok X Search, not web search and not the X Developer Platform MCP.", + inputSchema: xSearchInput, + outputSchema: webSearchOutput, + scopes: ["execute:search"], + auditLevel: "basic", + phase: 1, + sourceEndpoints: ["/v1/search"], +}; + // --- Tool 10: omniroute_web_fetch --- export const webFetchInput = z.object({ url: z @@ -1532,6 +1563,7 @@ export const MCP_TOOLS = [ listModelsCatalogTool, radarCatalogTool, webSearchTool, + xSearchTool, webFetchTool, simulateRouteTool, setBudgetGuardTool, diff --git a/open-sse/mcp-server/server.ts b/open-sse/mcp-server/server.ts index b2a866e9976..16e4eb8832e 100644 --- a/open-sse/mcp-server/server.ts +++ b/open-sse/mcp-server/server.ts @@ -18,6 +18,8 @@ import { costReportInput, listModelsCatalogInput, webSearchInput, + buildWebSearchInputSchema, + xSearchInput, webFetchInput, simulateRouteInput, setBudgetGuardInput, @@ -664,6 +666,28 @@ async function handleWebSearch(args: { } } +async function handleXSearch(args: { query: string; max_results?: number }) { + const start = Date.now(); + try { + const result = await omniRouteFetch("/v1/search", { + method: "POST", + body: JSON.stringify({ + query: args.query, + max_results: args.max_results ?? 5, + search_type: "x", + provider: "x-search", + }), + signal: AbortSignal.timeout(120000), + }); + await logToolCall("omniroute_x_search", args, result, Date.now() - start, true); + return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; + } catch (err) { + const msg = err instanceof Error ? err.message : String(err); + await logToolCall("omniroute_x_search", args, null, Date.now() - start, false, msg); + return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; + } +} + async function handleWebFetch(args: { url: string; provider?: "firecrawl" | "jina-reader" | "tavily-search" | "tinyfish"; @@ -697,7 +721,24 @@ async function handleWebFetch(args: { } } -export function createMcpServer(): McpServer { +export interface CreateMcpServerOptions { + blockedProviders?: string[] | (() => string[]); +} + +export function createMcpServer(options?: CreateMcpServerOptions): McpServer { + const resolveBlockedProviders = (): string[] => { + if (typeof options?.blockedProviders === "function") { + return options.blockedProviders(); + } + if (Array.isArray(options?.blockedProviders)) { + return options.blockedProviders; + } + return []; + }; + + const blockedProviders = resolveBlockedProviders(); + const dynamicWebSearchInput = buildWebSearchInputSchema(blockedProviders); + const server = new McpServer({ name: "omniroute", version: process.env.npm_package_version || "1.8.1", @@ -1021,13 +1062,27 @@ export function createMcpServer(): McpServer { { description: "Performs a web search using OmniRoute's search gateway. Supports multiple providers (Serper, Brave, Perplexity, Exa, Tavily) with automatic failover. Returns search results with titles, URLs, snippets, and position data.", - inputSchema: webSearchInput, + inputSchema: dynamicWebSearchInput, }, withScopeEnforcement("omniroute_web_search", (args) => - handleWebSearch(webSearchInput.parse(args)) + // Resolve per invocation (not the startup snapshot above) so a resolver + // function passed via CreateMcpServerOptions sees policy changes without + // a server rebuild. The advertised inputSchema stays a creation-time + // snapshot — MCP clients fetch it once at tools/list. + handleWebSearch(buildWebSearchInputSchema(resolveBlockedProviders()).parse(args)) ) ); + server.registerTool( + "omniroute_x_search", + { + description: + "Search X (Twitter) through OmniRoute using SuperGrok / xAI server-side x_search. Requires xai-oauth or an xAI API key. Not web search.", + inputSchema: xSearchInput, + }, + withScopeEnforcement("omniroute_x_search", (args) => handleXSearch(xSearchInput.parse(args))) + ); + server.registerTool( "omniroute_web_fetch", { diff --git a/open-sse/mcp-server/tools/memoryTools.ts b/open-sse/mcp-server/tools/memoryTools.ts index 16c835fd8e5..12a8c99c0d3 100644 --- a/open-sse/mcp-server/tools/memoryTools.ts +++ b/open-sse/mcp-server/tools/memoryTools.ts @@ -10,16 +10,24 @@ import { import { resolveMcpCallerApiKeyId } from "../mcpCallerIdentity.ts"; /** - * Resolve the memory owner id for an MCP tool call: - * explicit arg wins, otherwise fall back to the authenticated caller's - * principal id (HTTP auth headers on SSE/Streamable HTTP transports, - * OMNIROUTE_API_KEY env var on stdio). Keeps MCP-stored memories under - * the same owner id that chat-context memory uses, so retrieval in the - * chat pipeline finds entries written via MCP. + * Resolve the memory owner id for an MCP tool call. + * + * The authenticated caller's principal ALWAYS wins over a caller-supplied + * `apiKeyId` — otherwise any MCP caller could read, write, or delete another + * principal's memories by putting a different id in the tool arguments + * (GHSA-cpv3-xr7r-xf8q, IDOR). The caller is resolved from the per-request HTTP + * auth headers on SSE / Streamable HTTP transports, or from OMNIROUTE_API_KEY on + * stdio. The explicit argument is only honored as a fallback when no caller can + * be resolved (a bare local stdio process with no configured key — already + * trusted), preserving the local-tooling flow. Keeps MCP-stored memories under + * the same owner id that chat-context memory uses, so retrieval in the chat + * pipeline finds entries written via MCP. */ async function resolveMemoryOwnerId(explicit?: string): Promise { + const caller = await resolveMcpCallerApiKeyId().catch(() => undefined); + if (caller) return caller; if (explicit && explicit.trim() !== "") return explicit.trim(); - return (await resolveMcpCallerApiKeyId().catch(() => undefined)) || "mcp"; + return "mcp"; } export const MemorySearchSchema = z.object({ diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index d63c58e94ce..63407a38fcc 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -21,7 +21,11 @@ import { honorsRuleLockScope, } from "../config/providerErrorRules.ts"; import * as rot from "./rotationConfig.ts"; -import { getPassthroughProviders, getProviderCategory } from "../config/providerRegistry.ts"; +import { + getPassthroughProviders, + getProviderCategory, + isLocalProvider, +} from "../config/providerRegistry.ts"; import { DEFAULT_RESILIENCE_SETTINGS, resolveResilienceSettings, @@ -37,7 +41,12 @@ import { type FailureKind, } from "../../src/shared/utils/classify429"; import { recordProviderSuccess as resetCooldownFailureCount } from "./providerCooldownTracker.ts"; -import { resolveProviderId } from "../../src/shared/constants/providers"; +import { + getProviderById, + resolveProviderId, + isLocalProvider as isLocalProviderId, + isSelfHostedChatProvider, +} from "../../src/shared/constants/providers"; import { resolveUseUpstream429BreakerHints } from "../../src/shared/utils/providerHints"; import { getCodexModelScope } from "../config/codexQuotaScopes.ts"; import { getQuotaScopedModelForProvider } from "./antigravityQuotaFamily.ts"; @@ -791,12 +800,20 @@ export function hasPerModelQuota( return connectionPassthroughModels; } if (!provider) return false; - if (getCanonicalLockProvider(provider) === "antigravity") return true; - if (getCanonicalLockProvider(provider) === "codex") return true; - if (provider === "gemini" || provider === "github") return true; - if (provider === "antigravity" || provider === "agy") return true; - if (getPassthroughProviders().has(provider)) return true; - if (isCompatibleProvider(provider)) return true; + const canonicalId = resolveProviderId(provider); + if (getCanonicalLockProvider(canonicalId) === "antigravity") return true; + if (getCanonicalLockProvider(canonicalId) === "codex") return true; + if (canonicalId === "gemini" || canonicalId === "github") return true; + if (canonicalId === "antigravity" || canonicalId === "agy") return true; + if (getPassthroughProviders().has(canonicalId)) return true; + // #11071: getPassthroughProviders() reads the open-sse REGISTRY. A provider can declare + // passthroughModels:true in the SHARED registry (src/shared/constants/providers/) and be + // absent from that set — 40 of them are, and they are neither local nor self-hosted, so the + // branch below never reaches them either. Without this lookup a missing-model 404 on one of + // those cools the whole connection instead of locking out the single model. + if (getProviderById(canonicalId)?.passthroughModels === true) return true; + if (isCompatibleProvider(canonicalId)) return true; + if (isLocalProviderId(canonicalId) || isSelfHostedChatProvider(canonicalId)) return true; return false; } diff --git a/open-sse/services/autoCombo/freeAccessQuota.ts b/open-sse/services/autoCombo/freeAccessQuota.ts new file mode 100644 index 00000000000..515ec57a87f --- /dev/null +++ b/open-sse/services/autoCombo/freeAccessQuota.ts @@ -0,0 +1,209 @@ +/** + * Live wiring for STRICT_ZERO_COST's quota-based branch. + * + * Reuses the existing `getUsageForProvider()` (`open-sse/services/usage.ts`) + * instead of building a second quota system — this module only adds a short + * TTL cache in front of it (so a Telegram-scale request rate never triggers a + * live billing-API call per candidate per request) and an invalidation hook + * for the resilience layer to call the moment a 402/403/quota-exhausted + * response is observed (`accountFallback.ts`). + * + * The cache is intentionally synchronous to read: `resolveFreeAccessState()` + * never awaits. A cache miss returns `undefined` (→ UNKNOWN → excluded, + * fail-closed) and kicks off a background refresh for the *next* read — + * nothing here can make `strictZeroCostFilter.ts`'s pool build block on a + * network call. + */ +import { + getUsageForProvider, + USAGE_FETCHER_PROVIDERS, + type UsageFetcherProvider, +} from "./../usage.ts"; +import { getCachedProviderConnections } from "@/lib/db/readCache"; +import { defaultLogger as log } from "@omniroute/open-sse/utils/logger"; +import type { FreeAccessState } from "./strictZeroCostFilter"; + +const USAGE_FETCHER_PROVIDER_SET = new Set(USAGE_FETCHER_PROVIDERS); + +/** Default cache TTL, reused verbatim from the already-shipped + * `settings.autoRefreshProviderQuotaInterval` (180s default, + * `src/lib/db/settings.ts`) instead of inventing a new number. */ +const FALLBACK_TTL_MS = 180_000; + +/** Cold-cache thundering-herd guard: after a process restart every candidate + * in a pool build is a simultaneous cache miss, which without a cap would + * fire one `getUsageForProvider()` call per distinct (provider, connection) + * pair in the same tick. Capping concurrent background refreshes spreads + * that burst out — a skipped refresh here just means this candidate stays + * UNKNOWN (excluded, fail-closed) until a later pool build tries again, never + * a correctness issue. */ +const MAX_CONCURRENT_REFRESHES = 4; + +/** Entries older than this are pruned outright even if nothing ever triggers + * a fresh refresh for that exact key again (e.g. the connection was deleted + * and no candidate references it anymore, so a normal stale-triggered + * refresh — which self-heals inside one TTL window — never fires). A sweep, + * not a timer: piggybacks on `resolveFreeAccessState` calls so this module + * never owns its own background interval. */ +const HARD_EVICTION_AGE_MS = FALLBACK_TTL_MS * 20; // 1 hour at the default TTL +const SWEEP_EVERY_N_CALLS = 200; + +interface CacheEntry { + state: FreeAccessState; + fetchedAtMs: number; +} + +// Keyed by `${provider}::${connectionId}` — module-level, process-lifetime +// cache. Cleared per-entry by `invalidateFreeAccessState`, on a failed +// refresh, or by the periodic sweep below; never wholesale. +const cache = new Map(); +const inFlight = new Set(); +let resolveCallCount = 0; + +function cacheKey(provider: string, connectionId: string): string { + return `${provider}::${connectionId}`; +} + +// `getSettings()` is async (DB-backed); reading it synchronously here isn't +// possible without changing `resolveFreeAccessState`'s synchronous contract. +// Using the fallback unconditionally is equivalent in practice: it's the same +// number as `settings.autoRefreshProviderQuotaInterval`'s own default +// (`src/lib/db/settings.ts`), and `strictZeroCostFilter.ts`'s own +// `maxStateAgeMs` (passed the real, live setting from `virtualFactory.ts`) +// is the check that actually gates staleness for the STRICT_ZERO_COST +// decision — this cache TTL only bounds how long a background refresh is +// skipped, a looser, non-safety-critical concern. +function ttlMs(): number { + return FALLBACK_TTL_MS; +} + +/** Opportunistic sweep of very stale entries, run every N calls instead of on + * a timer. Cheap (a single Map iteration) and only ever removes entries no + * live candidate can plausibly still be waiting on. */ +function sweepIfDue(): void { + resolveCallCount += 1; + if (resolveCallCount % SWEEP_EVERY_N_CALLS !== 0) return; + const now = Date.now(); + for (const [key, entry] of cache) { + if (now - entry.fetchedAtMs > HARD_EVICTION_AGE_MS) cache.delete(key); + } +} + +/** + * Best-effort, provider-agnostic extraction of "how much free allowance is + * left" from whatever shape `getUsageForProvider()` returns for this + * provider today. Adapters were written for a human-readable quota display, + * not for this filter, so their payloads are heterogeneous; this function + * recognizes the two shapes already used by other read paths in this + * codebase (`quotas.*.remainingPercentage` / `.total`+`.remaining`, mirroring + * `quota_omniroute.py`'s own parsing) and returns `null` — never a guess — + * for anything else. `null` is treated as "not proven safe" by the filter. + */ +function extractRemainingAllowance(usage: unknown): number | null { + if (!usage || typeof usage !== "object") return null; + const quotas = (usage as Record).quotas; + if (!quotas || typeof quotas !== "object") return null; + + let worstPercent: number | null = null; + for (const raw of Object.values(quotas as Record)) { + if (!raw || typeof raw !== "object") continue; + const q = raw as Record; + if (q.unlimited === true) continue; + let pct: number | null = + typeof q.remainingPercentage === "number" ? q.remainingPercentage : null; + if ( + pct === null && + typeof q.total === "number" && + typeof q.remaining === "number" && + q.total > 0 + ) { + pct = (100 * q.remaining) / q.total; + } + if (pct === null) continue; + worstPercent = worstPercent === null ? pct : Math.min(worstPercent, pct); + } + return worstPercent; // percentage points; the filter's threshold is compared against this unit +} + +async function refresh(provider: string, connectionId: string): Promise { + const key = cacheKey(provider, connectionId); + if (inFlight.has(key)) return; + if (inFlight.size >= MAX_CONCURRENT_REFRESHES) return; // thundering-herd guard — see const doc above + inFlight.add(key); + try { + const connections = await getCachedProviderConnections(); + const connection = connections.find( + (c): c is Record => + !!c && + typeof c === "object" && + (c as Record).id === connectionId && + (c as Record).provider === provider + ); + if (!connection) { + cache.delete(key); + return; + } + const usage = await getUsageForProvider( + connection as unknown as Parameters[0], + { forceRefresh: false } + ); + const remaining = extractRemainingAllowance(usage); + const state: FreeAccessState = { + status: remaining === null ? "UNKNOWN" : remaining > 0 ? "SAFE" : "EXHAUSTED", + remainingFreeAllowance: remaining, + resetAt: + usage && + typeof usage === "object" && + typeof (usage as Record).resetAt === "string" + ? ((usage as Record).resetAt as string) + : null, + checkedAt: new Date().toISOString(), + }; + cache.set(key, { state, fetchedAtMs: Date.now() }); + } catch (err) { + // A failed lookup must never leave a stale SAFE entry behind — drop it so + // the next read is a clean cache miss (UNKNOWN), not a lucky reuse. + cache.delete(key); + log.warn("AUTO", "STRICT_ZERO_COST: usage refresh failed, treating as UNKNOWN", { + provider, + err: err instanceof Error ? err.message : String(err), + }); + } finally { + inFlight.delete(key); + } +} + +/** + * Synchronous read for `strictZeroCostFilter.ts`. Returns `undefined` when + * there's no usage adapter for this provider at all (a permanent UNKNOWN, no + * point ever refreshing), or on a cold/stale cache — in both cases a + * background refresh is kicked off (fire-and-forget, subject to the + * concurrency cap above) so a *later* read can benefit, but this call itself + * never blocks or throws. + */ +export function resolveFreeAccessState( + provider: string, + connectionId: string | undefined +): FreeAccessState | undefined { + sweepIfDue(); + if (!USAGE_FETCHER_PROVIDER_SET.has(provider as UsageFetcherProvider)) return undefined; + if (!connectionId) return undefined; + + const key = cacheKey(provider, connectionId); + const entry = cache.get(key); + const fresh = entry && Date.now() - entry.fetchedAtMs <= ttlMs(); + if (!fresh) { + void refresh(provider, connectionId); + } + return fresh ? entry.state : undefined; +} + +/** Called by `accountFallback.ts` the moment a 402/403/quota-exhausted + * response is classified for a connection — drops the cached entry + * immediately instead of waiting out the TTL, so the very next candidate-pool + * build reads a clean cache miss (UNKNOWN) rather than a stale SAFE. */ +export function invalidateFreeAccessState(provider: string, connectionId: string): void { + cache.delete(cacheKey(provider, connectionId)); +} + +export const __testing = { cache, extractRemainingAllowance, sweepIfDue }; diff --git a/open-sse/services/autoCombo/strictZeroCostFilter.ts b/open-sse/services/autoCombo/strictZeroCostFilter.ts new file mode 100644 index 00000000000..c9bc601bc05 --- /dev/null +++ b/open-sse/services/autoCombo/strictZeroCostFilter.ts @@ -0,0 +1,283 @@ +/** + * STRICT_ZERO_COST — an opt-in, stricter sibling of `hidePaidModels` + * (`paidModelFilter.ts`) for operators who need a hard guarantee against ANY + * incremental monetary spend, not just "documented as free". + * + * `hidePaidModels` answers "is this model classified free in FREE_MODEL_BUDGETS + * right now?" — a point-in-time catalog fact. It says nothing about whether a + * `recurring-*`/`one-time-initial` candidate's allowance has since been + * consumed, and nothing about whether exceeding it is a hard stop or silent + * pay-as-you-go billing. STRICT_ZERO_COST adds exactly those two checks, + * before ranking, before dispatch — never after. + * + * Design, kept deliberately close to `filterPaidOnlyCandidates`'s own stated + * goal: "a pure, dependency-light function so the filter is unit-testable in + * isolation". The live quota lookup (`getUsageForProvider`, cached with a TTL) + * lives in `freeAccessQuota.ts` and is injected here as a plain function — + * this file never imports the DB or makes a network call itself. + * + * No provider or model name appears anywhere in this file. A candidate passes + * or fails purely on the metadata it carries (`freeType`, `tos`, + * `hardStopGuaranteed`) plus, for quota-based types, a `FreeAccessState` + * resolved elsewhere. A future provider that ships correct metadata is + * handled automatically; one that doesn't is excluded automatically — see + * `docs/routing/STRICT_ZERO_COST.md`. + * + * ## Connection safety (fixed after code review, see `docs/routing/STRICT_ZERO_COST.md`) + * + * A candidate from `virtualFactory.ts`'s connection-based pool represents ONE + * provider/model pair with a set of *eligible* connections + * (`allowedConnectionIds`) — the actual connection used at dispatch is chosen + * later (session stickiness/LKGP), not by this filter. Two invariants follow: + * + * 1. The `keyless` shortcut (no live check needed, because no credential + * exists) is valid ONLY for candidates that genuinely came from the + * no-auth path — identified by `connectionId === SYNTHETIC_NOAUTH_CONNECTION_ID` + * (`resilienceCandidateFilter.ts`). A `keyless`-catalogued model reached + * through a real DB connection (the same provider also has a + * credentialed connection) does NOT get the shortcut — it falls through + * to the normal quota-based check like any other freeType, and is + * excluded unless that specific connection independently proves SAFE. + * 2. For a multi-account candidate (`connectionId: null`, + * `allowedConnectionIds: [...]`), each connection is checked + * INDIVIDUALLY. The returned candidate's `allowedConnectionIds` is + * REWRITTEN to exactly the subset proven SAFE — never the full original + * list. `autoStrategy.ts` (`open-sse/services/combo/autoStrategy.ts:315-331`) + * already intersects further routing against `allowedConnectionIds` + * before connection selection, so rewriting it here is enough to make + * "verified this connection" and "dispatch used this connection" the + * same set, by construction — no new enforcement point needed. + */ +import { + FREE_MODEL_BUDGETS, + type FreeModelBudget, +} from "@omniroute/open-sse/config/freeModelCatalog.ts"; +import { SYNTHETIC_NOAUTH_CONNECTION_ID } from "./resilienceCandidateFilter"; + +/** Types whose allowance needs no runtime verification: no credential exists + * for the candidate at all, so no request against it can ever be billed. */ +const KEYLESS_FREE_TYPES = new Set(["keyless"]); + +export type FreeAccessStatus = "SAFE" | "EXHAUSTED" | "UNKNOWN"; + +/** Live-checked allowance state for one (provider, connection) pair. Resolved + * and cached by `freeAccessQuota.ts`; passed in here as plain data so this + * module stays free of DB/network dependencies. */ +export interface FreeAccessState { + status: FreeAccessStatus; + /** Remaining free allowance in the provider's own unit (tokens, requests, or + * USD-equivalent) — whatever `getUsageForProvider()` reports. `null` when + * the provider's usage payload doesn't expose a numeric remaining figure. */ + remainingFreeAllowance: number | null; + /** When the allowance next resets, if the provider reports it. */ + resetAt: string | null; + /** When this state was fetched (ISO 8601) — used to detect staleness. */ + checkedAt: string; +} + +/** A candidate as this module needs to see it — a structural subset of + * `VirtualAutoComboCandidate` (`virtualFactory.ts`) so this file has no + * dependency on that module's full type. */ +export interface StrictZeroCostCandidate { + provider: string; + model: string; + connectionId: string | null; + allowedConnectionIds?: string[]; +} + +export interface StrictZeroCostOptions { + /** Master switch — mirrors `hidePaidModels`'s own off-by-default shape. */ + enabled: boolean; + /** + * Resolves the live allowance state for ONE specific (provider, connection) + * pair. Returns `undefined` when no usage capability exists for the + * provider at all (no adapter registered in `USAGE_FETCHER_PROVIDERS`), or + * when the cache has nothing fresh for this exact connection — both are a + * meaningful, terminal UNKNOWN for that connection, not an error to retry. + * + * Synchronous by design: the caller (`virtualFactory.ts`) resolves and + * caches state per candidate up front, once per pool build, so this filter + * itself never awaits a network call and stays trivially testable. + */ + resolveFreeAccessState: (provider: string, connectionId: string) => FreeAccessState | undefined; + /** Minimum remaining allowance (in the unit `resolveFreeAccessState` reports + * — percentage points for the built-in `freeAccessQuota.ts` resolver) a + * quota-based connection must exceed to pass. Must be >= 0; a fully-exhausted + * account (`remainingFreeAllowance === 0`) fails at any non-negative + * threshold via the strict `>` comparison below. */ + minRemainingAllowance: number; + /** Maximum age, in ms, a `FreeAccessState.checkedAt` may have before it's + * treated as stale (→ UNKNOWN, excluded). */ + maxStateAgeMs: number; + /** `now` injection for deterministic tests; defaults to `Date.now`. */ + now?: () => number; + /** + * The free-model catalog to look candidates up against. Defaults to the + * real, live `FREE_MODEL_BUDGETS` — overridable so tests can prove the + * autodiscovery contract (a provider/model that appears in the catalog is + * automatically considered; one that's removed automatically disappears) + * with synthetic fixtures instead of mutating global state. Production + * callers should never pass this. Threaded through by + * `filterStrictZeroCostCandidates` (previously accepted but silently + * ignored — fixed alongside the connection-safety review). + */ + catalog?: readonly FreeModelBudget[]; +} + +export function findBudgetEntry( + candidate: Pick, + catalog: readonly FreeModelBudget[] = FREE_MODEL_BUDGETS +): FreeModelBudget | undefined { + return catalog.find((m) => m.provider === candidate.provider && m.modelId === candidate.model); +} + +function isConnectionStateSafe( + provider: string, + connectionId: string, + resolveFreeAccessState: StrictZeroCostOptions["resolveFreeAccessState"], + options: Pick +): boolean { + const state = resolveFreeAccessState(provider, connectionId); + if (!state) return false; // no usage adapter for this provider, or lookup never ran/is stale + if (state.status !== "SAFE") return false; + + const now = (options.now ?? Date.now)(); + const checkedAtMs = Date.parse(state.checkedAt); + if (!Number.isFinite(checkedAtMs) || now - checkedAtMs > options.maxStateAgeMs) return false; + + if (state.remainingFreeAllowance === null) return false; + // A negative threshold would let a negative/garbage reading pass; a caller + // that genuinely wants "any allowance greater than zero" should pass 0. + if (options.minRemainingAllowance < 0) return false; + return state.remainingFreeAllowance > options.minRemainingAllowance; +} + +/** + * Decide which of a candidate's connections satisfy STRICT_ZERO_COST. Pure — + * `resolveFreeAccessState` is the only injected side-effecting dependency, + * and it's a synchronous cache read (see `StrictZeroCostOptions` above). + * + * Returns the list of connection ids proven SAFE right now: + * - `[SYNTHETIC_NOAUTH_CONNECTION_ID]` for a genuine no-auth candidate whose + * catalog entry is `keyless` — no live check needed or possible. + * - a (possibly empty) subset of the candidate's real connection id(s) for + * every other case, each individually verified. + * An empty array means the caller must exclude the candidate entirely. + */ +export function evaluateCandidateConnections( + candidate: StrictZeroCostCandidate, + budgetEntry: FreeModelBudget | undefined, + resolveFreeAccessState: StrictZeroCostOptions["resolveFreeAccessState"], + options: Pick +): string[] { + if (!budgetEntry) return []; // not in the catalog at all → paid, or genuinely unknown + + const isGenuineNoAuthCandidate = candidate.connectionId === SYNTHETIC_NOAUTH_CONNECTION_ID; + if (KEYLESS_FREE_TYPES.has(budgetEntry.freeType)) { + // The keyless shortcut is trustworthy ONLY when this specific candidate + // instance actually has no credential behind it. A `keyless`-catalogued + // model reached through a real DB connection (connectionId is a real id, + // or the candidate carries allowedConnectionIds at all) must NOT take + // this shortcut — it falls through to the quota-based check below like + // any other freeType, and is excluded there unless hardStopGuaranteed is + // also set for it (which the curated catalog does not do for keyless + // entries today, so it will correctly exclude). + if (isGenuineNoAuthCandidate) return [SYNTHETIC_NOAUTH_CONNECTION_ID]; + } + if (budgetEntry.freeType === "discontinued") return []; + if (isGenuineNoAuthCandidate) return []; // no-auth path but a non-keyless catalog entry: contradictory metadata, fail closed + + // Every remaining freeType (recurring-*, one-time-initial, a keyless entry + // reached via a real connection, and any future type this module doesn't + // special-case) requires a documented hard stop before any live check even + // runs — no point burning a quota lookup on a connection we could never + // trust regardless of its answer. + if (budgetEntry.hardStopGuaranteed !== true) return []; + + const candidateConnectionIds = candidate.connectionId + ? [candidate.connectionId] + : (candidate.allowedConnectionIds ?? []); + + const safe: string[] = []; + for (const connectionId of candidateConnectionIds) { + if (connectionId === SYNTHETIC_NOAUTH_CONNECTION_ID) continue; // never reachable here, defensive + if (isConnectionStateSafe(candidate.provider, connectionId, resolveFreeAccessState, options)) { + safe.push(connectionId); + } + } + return safe; +} + +/** + * Pool-level filter, same off-by-default identity contract as + * `filterPaidOnlyCandidates`. For a candidate that survives with a NARROWED + * connection set (the multi-account case), the returned object has + * `allowedConnectionIds` rewritten to exactly the SAFE subset — dispatch can + * then never select a connection this filter didn't verify, because + * `autoStrategy.ts` already enforces `allowedConnectionIds` as a hard + * allowlist downstream (see the module docstring above). + */ +export function filterStrictZeroCostCandidates( + pool: T[], + options: StrictZeroCostOptions +): T[] { + if (!options.enabled) return pool; + + const kept: T[] = []; + let changed = false; + for (const candidate of pool) { + const budgetEntry = findBudgetEntry(candidate, options.catalog); + const safeConnectionIds = evaluateCandidateConnections( + candidate, + budgetEntry, + options.resolveFreeAccessState, + options + ); + if (safeConnectionIds.length === 0) { + changed = true; + continue; + } + + const isGenuineNoAuthCandidate = candidate.connectionId === SYNTHETIC_NOAUTH_CONNECTION_ID; + const isSingleConnectionCandidate = candidate.connectionId !== null; + if (isGenuineNoAuthCandidate || isSingleConnectionCandidate) { + // Nothing to narrow — either the no-auth sentinel, or a candidate that + // already pointed at exactly one connection which proved safe. + kept.push(candidate); + continue; + } + + // Multi-account candidate: only rewrite if the safe subset is actually + // narrower than what was there before, to preserve the same + // identity-when-nothing-changed contract as `filterPaidOnlyCandidates`. + const original = candidate.allowedConnectionIds ?? []; + const isSameSet = + original.length === safeConnectionIds.length && + safeConnectionIds.every((id) => original.includes(id)); + if (isSameSet) { + kept.push(candidate); + } else { + changed = true; + kept.push({ ...candidate, allowedConnectionIds: safeConnectionIds }); + } + } + return changed ? kept : pool; +} + +/** + * Separate, optional ToS guard — kept independent from economic safety on + * purpose (Marco's requirement): a model can be economically SAFE and still + * excluded here for ToS reasons, or left in when this guard is off even if + * STRICT_ZERO_COST is on. Reuses the same curated `tos` field, no new data. + */ +export function filterTosAvoidCandidates( + pool: T[], + excludeTosAvoid: boolean, + catalog?: readonly FreeModelBudget[] +): T[] { + if (!excludeTosAvoid) return pool; + return pool.filter((candidate) => { + const budgetEntry = findBudgetEntry(candidate, catalog); + return budgetEntry?.tos !== "avoid"; + }); +} diff --git a/open-sse/services/autoCombo/virtualFactory.ts b/open-sse/services/autoCombo/virtualFactory.ts index 2eff8257ea1..28071144fbb 100644 --- a/open-sse/services/autoCombo/virtualFactory.ts +++ b/open-sse/services/autoCombo/virtualFactory.ts @@ -26,7 +26,10 @@ import { buildFamilyCandidateFilter, type ModelFamily } from "./modelFamily"; import { getHiddenModelsByProvider } from "@/models"; import { getSyncedAvailableModelsByConnection, getCustomModels } from "@/lib/db/models"; import { filterPaidOnlyCandidates } from "./paidModelFilter"; +import { filterStrictZeroCostCandidates, filterTosAvoidCandidates } from "./strictZeroCostFilter"; +import { resolveFreeAccessState } from "./freeAccessQuota"; import { isModelExcludedByConnection } from "@/domain/connectionModelRules"; +import { resolveProviderAlias } from "../model.ts"; import { filterExcludedCandidates } from "./candidateOverrides"; import { getExcludedConnectionIds } from "@/lib/db/autoCandidateOverrides"; import { @@ -273,9 +276,19 @@ function getNoAuthCandidates( // modelCompatOverrides/customModels key_value namespaces) the same way the // credentialed-connection loop below does, so a hidden no-auth model never // enters the auto-combo/fusion candidate pool either. - const hiddenModels = - hiddenModelsMap.get(providerId) ?? - (typeof providerDef.alias === "string" ? hiddenModelsMap.get(providerDef.alias) : undefined); + const hiddenLookupIds = [ + providerId, + typeof providerDef.alias === "string" ? providerDef.alias : null, + registryAlias, + routingPrefix, + resolveProviderAlias(providerId), + resolveProviderAlias(routingPrefix), + ]; + const hiddenModels = new Set(); + for (const id of hiddenLookupIds) { + if (!id) continue; + for (const modelId of hiddenModelsMap.get(id) ?? []) hiddenModels.add(modelId); + } for (const model of registryModels) { const modelId = typeof model?.id === "string" && model.id.trim().length > 0 ? model.id : null; @@ -579,6 +592,29 @@ export async function prepareVirtualAutoComboInputs( // exclude paid-only backends from EVERY `auto/*` candidate pool. const paidFilteredPool = filterPaidOnlyCandidates(pool, settings.hidePaidModels === true); if (paidFilteredPool !== pool) pool = paidFilteredPool; + + // STRICT_ZERO_COST: opt-in, off by default (`settings.freeAccessPolicy !== "strict"` + // leaves `pool` byte-identical, same contract as `hidePaidModels`). See + // `strictZeroCostFilter.ts` for why this is stricter than `hidePaidModels` alone — + // including the connection-safety invariant it enforces per-connection, not just + // per-candidate: `resolveFreeAccessState` here is a raw pass-through of the real + // per-(provider,connectionId) resolver; the filter itself decides which connection(s) + // on each candidate to check and rewrites `allowedConnectionIds` to the SAFE subset. + const strictFilteredPool = filterStrictZeroCostCandidates(pool, { + enabled: settings.freeAccessPolicy === "strict", + resolveFreeAccessState, + // 1 percentage point of headroom, not 0: `freeAccessQuota.ts` reports + // remaining allowance as a percentage, and a raw ">0" comparison would + // let a reading of e.g. 0.3% (rounding noise, not real headroom) pass. + minRemainingAllowance: 1, + maxStateAgeMs: (Number(settings.autoRefreshProviderQuotaInterval) || 180) * 1000, + }); + if (strictFilteredPool !== pool) pool = strictFilteredPool; + + // Separate, optional ToS guard — independent of economic safety on purpose. + const tosFilteredPool = filterTosAvoidCandidates(pool, settings.excludeTosAvoid === true); + if (tosFilteredPool !== pool) pool = tosFilteredPool; + return pool; }; diff --git a/open-sse/services/browserBackedChat.ts b/open-sse/services/browserBackedChat.ts index 848f5642582..4b3c7078e7b 100644 --- a/open-sse/services/browserBackedChat.ts +++ b/open-sse/services/browserBackedChat.ts @@ -275,11 +275,13 @@ export async function browserBackedChat( }); try { const tNavStart = Date.now(); - await page.goto(chatPageUrl, { - waitUntil: "domcontentloaded", - timeout: 60000, - signal: signal ?? undefined, - }); + await withAbort( + page.goto(chatPageUrl, { + waitUntil: "domcontentloaded", + timeout: 60000, + }), + signal + ); await waitWithSignal(2500, signal); const navigateMs = Date.now() - tNavStart; @@ -617,11 +619,13 @@ async function doCookieRefreshOnContext( ): Promise { const page = await openPage(pooled); try { - await page.goto(chatPageUrl, { - waitUntil: "domcontentloaded", - timeout: 60000, - signal: signal ?? undefined, - }); + await withAbort( + page.goto(chatPageUrl, { + waitUntil: "domcontentloaded", + timeout: 60000, + }), + signal + ); return await waitForCookiesWithPolling(pooled.context, cookieDomain, signal); } catch (err) { if (err instanceof DOMException && err.name === "AbortError") throw err; diff --git a/open-sse/services/claudeCodeToolRemapper.ts b/open-sse/services/claudeCodeToolRemapper.ts index 15995a95003..82e408e7c87 100644 --- a/open-sse/services/claudeCodeToolRemapper.ts +++ b/open-sse/services/claudeCodeToolRemapper.ts @@ -57,6 +57,10 @@ const TOOL_RENAME_MAP: Record = { cronlist: "CronList", taskoutput: "TaskOutput", taskstop: "TaskStop", + taskcreate: "TaskCreate", + taskupdate: "TaskUpdate", + tasklist: "TaskList", + taskget: "TaskGet", workflow: "Workflow", }; @@ -205,14 +209,22 @@ export function remapToolNamesInResponse( * Restore a tool name for Claude-format clients (#9008). * * Preference order: - * 1. Exact `_toolNameMap` hit (sanitized → original) - * 2. Case-insensitive match against map keys/values (Gemini/Antigravity may - * echo a lowercased name for a PascalCase Claude Code tool) - * 3. REVERSE_MAP TitleCase → lowercase fallback for clients with no request map - * (#7926 XML / OpenCode-style lowercase tools) + * 1. Exact `_toolNameMap` hit where the value differs from the key + * (sanitized → original request-side alias) + * 2. Canonical casing upgrade for known Claude Code tools + * (`croncreate` → `CronCreate`, `bash` → `Bash`, …) + * 3. Case-insensitive non-identity match against map keys/values + * (Gemini/Antigravity may echo a lowercased name for a PascalCase + * Claude Code tool) + * 4. Identity echo kept ONLY when no canonical upgrade exists + * 5. No-map fallbacks: REVERSE_MAP TitleCase → lowercase (#7926 XML / + * OpenCode-style lowercase tools), then the static table * - * Never apply REVERSE_MAP after a request-side original is known — that is what - * turned Claude Code's `Read`/`WebSearch` into `read`/`websearch`. + * Identity entries (key === value) never pin a known tool below its + * canonical casing. Some upstream gateways echo the very lowercase name + * they emitted into the alias channel; honouring that echo is what let a + * literal `croncreate` reach Claude Code as an unknown tool even though + * the request declared `CronCreate`. */ export function restoreClaudeToolName( rawName: string, @@ -220,27 +232,49 @@ export function restoreClaudeToolName( ): string { if (!rawName) return rawName; - const exact = toolNameMap?.get(rawName); - if (typeof exact === "string") return exact; + // Undefined when rawName already IS the canonical form — an input that + // maps to itself must keep flowing to the #7926 legacy paths below. + const lower = rawName.toLowerCase(); + const canonicalRaw = TOOL_RENAME_MAP[lower]; + const canonical = canonicalRaw && canonicalRaw !== rawName ? canonicalRaw : undefined; if (toolNameMap?.size) { - const lower = rawName.toLowerCase(); + const exact = toolNameMap.get(rawName); + if (typeof exact === "string" && (exact !== rawName || !canonical)) { + return exact; + } + + let identityMatch: string | undefined; for (const [sanitized, original] of toolNameMap.entries()) { - if (sanitized.toLowerCase() === lower || original.toLowerCase() === lower) { + if (sanitized.toLowerCase() !== lower && original.toLowerCase() !== lower) { + continue; + } + if (original !== rawName) { return original; } + identityMatch = original; + } + if (identityMatch !== undefined && !canonical) { + return identityMatch; } } + // Canonical echo is terminal: when the upstream echoes back the exact + // canonical form the request declared, keep it verbatim. The #7926 + // REVERSE_MAP fallbacks below would otherwise downcase it for routes that + // carry no _toolNameMap (Claude Code → OpenAI-style upstreams), which is + // what let a literal `croncreate` reach Claude Code even though the client + // declared `CronCreate` (live repro, PR #11085). + if (canonicalRaw === rawName) return rawName; + + if (canonical) return canonical; + // When no request toolNameMap is provided (e.g. non-Claude client): // If rawName is already TitleCase, apply REVERSE_MAP for #7926 backward compatibility (Bash → bash). if (!toolNameMap && REVERSE_MAP[rawName]) { return REVERSE_MAP[rawName]; } - const canonical = TOOL_RENAME_MAP[rawName.toLowerCase()]; - if (canonical) return canonical; - return REVERSE_MAP[rawName] ?? rawName; } diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index f01d206ecab..2d83935794b 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -34,6 +34,7 @@ import { recordComboFailure, } from "./combo/failureTracker.ts"; import { buildNoUpstreamResponseDiagnostics, buildRecoveryHint } from "./combo/pinRecovery.ts"; +import { formatExhaustedConnectionKey } from "./combo/comboDiagFormat.ts"; import { buildTargetTimeoutRunner } from "./combo/targetTimeoutRunner.ts"; import { recordComboRequest, recordComboShadowRequest, getComboMetrics } from "./comboMetrics.ts"; import { qualityScoreFor } from "./routing/index.ts"; @@ -1081,10 +1082,7 @@ async function handleComboChatInner({ attempted: recordedAttempts, excluded: [ ...[...exhaustedProviders].map((p) => ({ provider: p, reason: "exhausted" })), - ...[...exhaustedConnections].map((c) => ({ - provider: "unknown", - reason: `exhausted_connection:${String(c).slice(0, 8)}`, - })), + ...[...exhaustedConnections].map((c) => formatExhaustedConnectionKey(String(c))), ], attemptOrder: comboAttemptOrder, terminalReason, @@ -2727,11 +2725,22 @@ async function handleComboChatInner({ ); } const retryAfterSeconds = undefined; + // #10966: when every observed failure was independently classified as quota/ + // balance exhaustion (isQuotaExhaustionResponse, tracked via observeFailure's + // allObservedFailuresQuota accumulator), stamp a stable `quota_exhausted` + // terminalReason instead of forwarding the raw upstream error string. The raw + // string falls through buildRecoveryHint's default branch ("retry" / "failed + // transiently"), which is actively misleading for a durable wallet/quota + // exhaustion — retrying the same combo will never refill it. + const terminalReason = + observedFailure && allObservedFailuresQuota + ? "quota_exhausted" + : (lastError ?? "all_models_failed"); return withQuotaExhaustionClassification( errorResponseWithComboDiagnostics( status, msg, - buildComboDiag(lastError ?? "all_models_failed", retryAfterSeconds) + buildComboDiag(terminalReason, retryAfterSeconds) ), observedFailure ? allObservedFailuresQuota : null ); diff --git a/open-sse/services/combo/comboDiagFormat.ts b/open-sse/services/combo/comboDiagFormat.ts new file mode 100644 index 00000000000..caed8dedfd9 --- /dev/null +++ b/open-sse/services/combo/comboDiagFormat.ts @@ -0,0 +1,29 @@ +/** + * #10967: format an `exhaustedConnections` key stored by targetExhaustion.ts + * (`markAuthLevelExhaustion` / `markAgentrouterConnectionQuotaExhaustion` / + * `markConnectionLevelExhaustion`, all keyed as `` `${provider}:${connectionId}` ``) + * into a diagnostics `excluded` entry. + * + * Before this fix, `buildComboDiag` (combo.ts) hardcoded `provider: "unknown"` and + * `slice(0, 8)`'d the WHOLE key — for a typical `jina-ai:` key that produced + * `exhausted_connection:jina-ai:` (the 7-char provider id + colon consumed the + * entire 8-char budget, the UUID silently dropped, and the real provider id + * discarded in favor of the literal string "unknown"). + * + * Splitting on the FIRST `:` recovers the real provider id and truncates only the + * connection-id half (never the full UUID, matching the public combo projection's + * connection-id redaction policy — #2300). + */ +export function formatExhaustedConnectionKey(key: string): { + provider: string; + reason: string; +} { + const raw = String(key); + const sepIdx = raw.indexOf(":"); + const provider = sepIdx >= 0 ? raw.slice(0, sepIdx) : ""; + const connId = sepIdx >= 0 ? raw.slice(sepIdx + 1) : raw; + return { + provider: provider || "unknown", + reason: `exhausted_connection:${connId.slice(0, 8)}`, + }; +} diff --git a/open-sse/services/combo/pinRecovery.ts b/open-sse/services/combo/pinRecovery.ts index d8cd728443e..6923eb5204b 100644 --- a/open-sse/services/combo/pinRecovery.ts +++ b/open-sse/services/combo/pinRecovery.ts @@ -32,6 +32,12 @@ export function buildRecoveryHint( next_step: "No active accounts are connected for this combo. Open /dashboard/providers, reconnect at least one, then retry.", }; + case "quota_exhausted": + return { + action: "switch-combo", + next_step: + "Every target in this combo failed with a quota or account-balance exhaustion error. Top up the account/wallet or switch to a combo/provider with available quota — this will not recover on retry.", + }; case "all_models_failed": return { action: "try-auto", diff --git a/open-sse/services/combo/quotaExhaustion.ts b/open-sse/services/combo/quotaExhaustion.ts index 50e2d71863f..78c3a8cf1d4 100644 --- a/open-sse/services/combo/quotaExhaustion.ts +++ b/open-sse/services/combo/quotaExhaustion.ts @@ -6,6 +6,10 @@ const TERMINAL_QUOTA_CODES = new Set([ "credits_exhausted", "insufficient_quota", "quota_exhausted", + // #10966: durable wallet/balance exhaustion signalled on a 403 (not 402/429) by + // some upstreams — e.g. AUTHZ_INSUFFICIENT_BALANCE, "Insufficient account balance. + // Top up your account at …". + "authz_insufficient_balance", ]); const trustedClassifications = new WeakMap(); @@ -62,7 +66,13 @@ export async function isQuotaExhaustionResponse( const trusted = trustedClassifications.get(response); if (trusted !== undefined) return trusted; - if (response.status !== 402 && response.status !== 429) return false; + // #10966: 403 is included alongside 402/429 — some upstreams (e.g. durable + // wallet/balance exhaustion) return a 403 for a terminal quota condition instead + // of the more common 402/429. The structured-code/text checks below still gate + // this to genuine quota signals (CREDITS_EXHAUSTED_SIGNALS / TERMINAL_QUOTA_CODES / + // checkFallbackError's own classification), so a generic auth-only 403 (invalid + // key, no matching quota signal) still falls through to `false`. + if (response.status !== 402 && response.status !== 429 && response.status !== 403) return false; const { text, structuredError } = await parseError(response); if (provider === "gemini" && response.status === 429) { diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index 7c7f0ce8a92..76ef5546aba 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -617,7 +617,18 @@ export async function validateResponseQuality( try { json = JSON.parse(text); } catch { - if (text.startsWith("data:") || text.startsWith("event:")) return { valid: true }; + // An SSE stream body is expected for streamed upstreams. Besides `data:` and + // `event:` frames, the SSE spec also allows comment lines that begin with a + // colon (`:`), which providers use for keep-alives while the model is still + // generating — e.g. OpenRouter emits `: OPENROUTER PROCESSING` on slower / + // reasoning responses. A stream that opens with such a comment (or with + // leading whitespace/newlines) is still a valid stream, not malformed JSON, + // so trim and recognize the comment prefix before rejecting. Without this, + // otherwise-good streamed completions get failed as "not valid JSON". + const trimmed = text.trimStart(); + if (trimmed.startsWith("data:") || trimmed.startsWith("event:") || trimmed.startsWith(":")) { + return { valid: true }; + } return { valid: false, reason: "response is not valid JSON" }; } diff --git a/open-sse/services/compression/engines/ccr/index.ts b/open-sse/services/compression/engines/ccr/index.ts index d2136fc8f19..92869bfbc9b 100644 --- a/open-sse/services/compression/engines/ccr/index.ts +++ b/open-sse/services/compression/engines/ccr/index.ts @@ -45,7 +45,7 @@ import { } from "../../../../../src/lib/db/ccrBlocks.ts"; import { createCompressionStats } from "../../stats.ts"; import { queryBlock, type CcrQuery } from "./ccrQuery.ts"; -import { injectCcrProtocolInstruction } from "./protocolInstruction.ts"; +import { callerSupportsCcrRetrieve, injectCcrProtocolInstruction } from "./protocolInstruction.ts"; import type { CompressionEngine, CompressionEngineApplyOptions, @@ -939,6 +939,30 @@ export const ccrEngine: CompressionEngine = { return { body, compressed: false, stats: null }; } + // #7746 follow-up: only callers whose tools[] proves they can reach + // omniroute_ccr_retrieve may have content replaced at all. For everyone + // else (plain OpenAI-compatible clients — the marker is an MCP-only + // contract) replacement would strand the original text behind a hash the + // model has no way to resolve. Skip the whole engine for them. The check + // is wrapped defensively: a malformed body must fail OPEN (no + // compression), never throw into the request pipeline. + let callerCanRetrieve = false; + try { + callerCanRetrieve = callerSupportsCcrRetrieve(body); + } catch (err) { + // Defensive: the helper is total, but if it ever throws we must fail + // OPEN (no compression) — and surface it so a future regression in the + // helper is visible instead of silently bypassing compression forever. + console.warn( + "[compression/ccr] callerSupportsCcrRetrieve threw; skipping compression:", + err instanceof Error ? err.message : err + ); + callerCanRetrieve = false; + } + if (!callerCanRetrieve) { + return { body, compressed: false, stats: null }; + } + const minChars = typeof stepConfig["minChars"] === "number" ? (stepConfig["minChars"] as number) diff --git a/open-sse/services/contextManager.ts b/open-sse/services/contextManager.ts index a2d678f107c..6fe9e94c8e2 100644 --- a/open-sse/services/contextManager.ts +++ b/open-sse/services/contextManager.ts @@ -669,13 +669,35 @@ function purifyHistory(messages: Record[], targetTokens: number result = fixToolPairs(result); result = stripTrailingAssistantOrphanToolUse(result); - // Add summary of dropped messages + // Add summary of dropped messages. Merge the notice INTO the leading + // system/developer message instead of splicing a second system-role message + // mid-array: strict gateways (TokenRouter confirmed live 2026-08-22, see the + // PROVIDERS_SYSTEM_MUST_BE_FIRST list in src/lib/memory/injection.ts) reject + // any system message at index > 0 with HTTP 400 "System message must be at + // the beginning". When there is no leading system message, prepend one -- + // index 0 is accepted by every provider (same slot the old splice used when + // system[] was empty). if (keep < nonSystem.length) { const dropped = nonSystem.length - keep; - result.splice(system.length, 0, { - role: "system", - content: `[Context compressed: ${dropped} earlier messages removed to fit context window]`, - }); + const droppedNotice = `[Context compressed: ${dropped} earlier messages removed to fit context window]`; + const first = result[0]; + if (first && (first.role === "system" || first.role === "developer")) { + if (typeof first.content === "string") { + result[0] = { + ...first, + content: first.content ? `${droppedNotice}\n${first.content}` : droppedNotice, + }; + } else if (Array.isArray(first.content)) { + result[0] = { + ...first, + content: [{ type: "text", text: droppedNotice }, ...(first.content as unknown[])], + }; + } else { + result[0] = { ...first, content: droppedNotice }; + } + } else { + result.unshift({ role: "system", content: droppedNotice }); + } } return result; diff --git a/open-sse/services/cursorApiKeyAuth.ts b/open-sse/services/cursorApiKeyAuth.ts index 60c3385783d..126123b119d 100644 --- a/open-sse/services/cursorApiKeyAuth.ts +++ b/open-sse/services/cursorApiKeyAuth.ts @@ -50,8 +50,12 @@ export function isCursorApiKey(value: unknown): value is string { return typeof value === "string" && value.startsWith(CURSOR_API_KEY_PREFIX); } +// Session-cache key fingerprint, not a password/credential hash — keyed with a fixed context +// label so it reads as a domain-separated digest rather than a bare password hash. function cacheKeyFor(apiKey: string): string { - return crypto.createHash("sha256").update(apiKey).digest("hex"); + return crypto.createHmac("sha256", "omniroute-cursor-session-cache-fingerprint-v1") + .update(apiKey) + .digest("hex"); } export function readJwtExpiryMs(token: string): number | null { diff --git a/open-sse/services/defaultReasoningEffort.ts b/open-sse/services/defaultReasoningEffort.ts index 156312a24c3..4c09643147a 100644 --- a/open-sse/services/defaultReasoningEffort.ts +++ b/open-sse/services/defaultReasoningEffort.ts @@ -30,18 +30,28 @@ function hasExplicitReasoningField(body: Record): boolean { * * `suffixEffort` (#7694) is the tier a `/-{effort}` synced-model alias * resolved to (`src/sse/services/model.ts`'s `resolveSyncedModelIdAndEffort`) — an - * explicit, request-time model selection, so it takes priority over the static - * `ModelSpec.defaultReasoningEffort` fleet-wide default (#6879) when both are present. + * explicit, request-time model selection, so it takes priority over both defaults + * below when present. + * + * `syncedDefaultEffort` is the vendor-declared default captured at sync time + * (`reasoning.default_effort`, e.g. OpenRouter `stealth/ox-alpha` declares `max`) — + * see `detectDefaultThinkingEffort`. A model that only produces usable output with + * an explicit effort gets the vendor default instead of an empty upstream response. + * It is the LOWEST-priority default: an explicit client value wins, the suffix alias + * wins, and a static `ModelSpec.defaultReasoningEffort` (operator-configured + * strip-by-default, #6879) also wins over the vendor default. */ export function applyDefaultReasoningEffort>( body: T, modelId: string, - suffixEffort?: string | null + suffixEffort?: string | null, + syncedDefaultEffort?: string | null ): T { if (!body || typeof body !== "object") return body; if (hasExplicitReasoningField(body)) return body; - const defaultEffort = suffixEffort || getModelSpec(modelId)?.defaultReasoningEffort; + const defaultEffort = + suffixEffort || getModelSpec(modelId)?.defaultReasoningEffort || syncedDefaultEffort; if (!defaultEffort) return body; return { ...body, reasoning_effort: defaultEffort }; diff --git a/open-sse/services/learnedReasoningEffortCaps.ts b/open-sse/services/learnedReasoningEffortCaps.ts new file mode 100644 index 00000000000..b0125d86837 --- /dev/null +++ b/open-sse/services/learnedReasoningEffortCaps.ts @@ -0,0 +1,126 @@ +/** + * Learned Reasoning-Effort Caps — reactive capability memory for providers/models + * OmniRoute has no static registry entry for (custom OpenAI-compatible connections, + * or any registered provider whose registry entry carries no reasoning metadata). + * + * Same shape as `learnedThinkingCaps.ts` (thinking_budget), generalized from a + * numeric budget to an ordinal reasoning_effort scale: on a 4xx whose body + * enumerates the accepted values, `base.ts`'s executor calls + * `recordLearnedReasoningEffort`, which stores the highest recognized value in a + * module-level Map keyed "provider:model" (lowercased). Subsequent requests for + * the same provider+model read the cap via `getLearnedReasoningEffort` (consulted + * by `sanitizeReasoningEffortForProvider` in `executors/base/reasoningEffort.ts`) + * so the 4xx→retry round-trip is paid at most once per process per provider+model. + * + * In-memory only (same operator-accepted tradeoff as the thinking-budget cache): + * restart resets, the first request after a restart may re-learn at the cost of + * one upstream 4xx. + */ + +export const REASONING_EFFORT_ORDER: readonly string[] = [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", +]; + +// key: `${provider}:${model}` lowercased → highest value known to be accepted. +const learnedCaps = new Map(); + +function buildKey(provider: string | null | undefined, model: string | null | undefined): string { + const p = typeof provider === "string" ? provider.trim().toLowerCase() : ""; + const m = typeof model === "string" ? model.trim().toLowerCase() : ""; + if (!p || !m) return ""; + return `${p}:${m}`; +} + +function rankOf(value: string): number { + return REASONING_EFFORT_ORDER.indexOf(value); +} + +/** + * Return the learned cap for provider+model, or null when nothing has been + * learned yet (no upstream 4xx recorded). Keyed case-insensitively. + */ +export function getLearnedReasoningEffort( + provider: string | null | undefined, + model: string | null | undefined +): string | null { + const key = buildKey(provider, model); + if (!key) return null; + return learnedCaps.get(key) ?? null; +} + +/** + * Record that `acceptedValues` is the enum the upstream advertised for + * provider+model, and store the highest recognized value as the learned cap. + * Returns the stored value, or null when `acceptedValues` contained no token + * from `REASONING_EFFORT_ORDER` (nothing usable to learn) or the key is unusable. + * + * Always monotonically decreases: if a cap already stored ranks lower than the + * newly computed highest, the stored (lower) value wins and is returned + * unchanged. This keeps a later, laxer-looking response (or a race between + * concurrent requests) from ratcheting the cap back up. + */ +export function recordLearnedReasoningEffort( + provider: string | null | undefined, + model: string | null | undefined, + acceptedValues: string[] +): string | null { + const key = buildKey(provider, model); + if (!key) return null; + + let best: string | null = null; + let bestRank = -1; + for (const raw of acceptedValues) { + const rank = rankOf(raw); + if (rank > bestRank) { + bestRank = rank; + best = raw; + } + } + if (best === null) return null; + + const existing = learnedCaps.get(key); + if (existing !== undefined && rankOf(existing) <= bestRank) { + return existing; // already learned an equal-or-lower cap; keep it + } + learnedCaps.set(key, best); + return best; +} + +// Matches both prose shapes observed: OVH's `@ai-sdk/openai-compatible` +// deserializer ("expected one of `a`, `b`") and a generic vendor prose form +// ("Supported types are a, b, and c"). +const LIST_INTRO = /(?:expected one of|supported (?:types|values) are)[:\s]*([^.]+)/i; + +/** + * Extract the upstream-advertised accepted reasoning_effort values from a 4xx + * error body. Returns only tokens present in REASONING_EFFORT_ORDER (unknown + * tokens are dropped defensively) in the order they appeared, or null when the + * text names no recognized enum member. + */ +export function parseReasoningEffortEnum(errText: unknown): string[] | null { + if (typeof errText !== "string" || !errText) return null; + const match = LIST_INTRO.exec(errText); + if (!match) return null; + const tokens = match[1] + .split(/,|\band\b|&/i) + .map((t) => + t + .replace(/`/g, "") + .replace(/\([^)]*\)/g, "") + .trim() + .toLowerCase() + ) + .filter((t) => t.length > 0 && REASONING_EFFORT_ORDER.includes(t)); + return tokens.length > 0 ? tokens : null; +} + +/** Test-only: clear the learned-cap Map between tests. */ +export function __test_resetLearnedReasoningEffortCaps(): void { + learnedCaps.clear(); +} diff --git a/open-sse/services/reasoningInputPolicy.ts b/open-sse/services/reasoningInputPolicy.ts index 71a283a706d..94f27ddcdc9 100644 --- a/open-sse/services/reasoningInputPolicy.ts +++ b/open-sse/services/reasoningInputPolicy.ts @@ -1,5 +1,6 @@ import { REGISTRY } from "../config/providerRegistry.ts"; import type { ReasoningTransport } from "../config/providerRegistry.ts"; +import { isValidResponsesItemId } from "./responsesItemId.ts"; type JsonRecord = Record; @@ -279,13 +280,27 @@ function sanitizeResponsesInput( if (!hasPlaintext && !hasOpaque && (!hasDisplaySummary(next) || stripOrphanedSummaries)) { continue; } - if (!hasOpaque && typeof next.id === "string") delete next.id; + // `id` is only worth keeping on an opaque item with a valid string value — + // non-opaque items don't replay their id, and a malformed value (e.g. `null`, + // observed on opencode/zen) must not survive either way (#11108). + if (!hasOpaque || !isValidResponsesItemId(next.id)) delete next.id; + // Some upstreams (e.g. opencode/zen) omit `summary` entirely on opaque + // reasoning items instead of sending an empty array. Replaying that shape + // verbatim trips strict Responses-API validators that require the field + // to be present on every `input[]` item of type `reasoning` (#11108). + // Plaintext-only items intentionally have no `summary` key and must stay + // untouched. + if (hasOpaque && next.summary === undefined) next.summary = []; filtered.push(next); continue; } const cloned = { ...record }; - if (typeof cloned.id === "string") delete cloned.id; + // Strip `id` whenever present, valid or not: these items don't need a + // replayed server id, and a malformed one (e.g. `null`, same opencode/zen + // omission pattern as the reasoning branch above) must not survive either + // (#11108). + if (cloned.id !== undefined) delete cloned.id; filtered.push(cloned); } return filtered; diff --git a/open-sse/services/responsesInputSanitizer.ts b/open-sse/services/responsesInputSanitizer.ts index 5d81b787a10..94cd99f934b 100644 --- a/open-sse/services/responsesInputSanitizer.ts +++ b/open-sse/services/responsesInputSanitizer.ts @@ -1,3 +1,5 @@ +import { isValidResponsesItemId } from "./responsesItemId.ts"; + type JsonRecord = Record; type SanitizeResponsesInputOptions = { dropInternalAssistantMessages?: boolean; @@ -40,7 +42,12 @@ function sanitizeFunctionName(name: string): string { } function sanitizeInputItemId(record: JsonRecord): JsonRecord { - if (typeof record.id !== "string") return record; + if (record.id === undefined) return record; + if (!isValidResponsesItemId(record.id)) { + const next = { ...record }; + delete next.id; + return next; + } const type = typeof record.type === "string" ? record.type : ""; const expectedPrefix = SERVER_ITEM_ID_PREFIX_BY_TYPE[type]; diff --git a/open-sse/services/responsesItemId.ts b/open-sse/services/responsesItemId.ts new file mode 100644 index 00000000000..a57ac92e42f --- /dev/null +++ b/open-sse/services/responsesItemId.ts @@ -0,0 +1,7 @@ +// Shared by reasoningInputPolicy.ts and responsesInputSanitizer.ts: both strip a +// Responses-API `input[]` item's `id` field when it isn't a valid string before +// replay, so a malformed value (e.g. `null`, observed on opencode/zen) never +// survives to trip a strict upstream with "Expected 'id' to be a string." (#11108). +export function isValidResponsesItemId(id: unknown): id is string { + return typeof id === "string"; +} diff --git a/open-sse/services/streamRecovery.ts b/open-sse/services/streamRecovery.ts index a0a397fd87b..3a95e9a2a3f 100644 --- a/open-sse/services/streamRecovery.ts +++ b/open-sse/services/streamRecovery.ts @@ -183,10 +183,33 @@ export function hasTerminalMarker(bytes: Uint8Array): boolean { export interface OpenAiSseScan { /** Concatenated assistant text seen across `choices[].delta.content`. */ text: string; + /** Concatenated reasoning trace seen across `choices[].delta.reasoning_content`. Some + * providers stream the entire answer here and leave `content` empty/null — tracked + * separately so a clean stop with reasoning-only output can still be recognized as + * "nothing usable was delivered" instead of "a normal empty turn". */ + reasoningText: string; /** True if any `choices[].delta.tool_calls` appeared — NEVER continue those. */ sawToolCall: boolean; - /** True if a terminal marker (`[DONE]` or a non-null `finish_reason`) appeared. */ + /** + * True only when `tool_calls` appeared in this scan AND its own + * `finish_reason: "tool_calls"` has NOT also appeared in the same scan — i.e. the + * call is still being streamed (arguments may be mid-flight). Once + * `finish_reason: "tool_calls"` closes it, the call is complete, not in flight: the + * client has the full arguments and a truncation past this point only drops + * trailing prose, which continuation can safely recover. + */ + sawToolCallInFlight: boolean; + /** + * True if a terminal marker for the OVERALL stream appeared: `[DONE]`, or a + * `finish_reason` other than `"tool_calls"`. A `finish_reason: "tool_calls"` ends + * that one choice but is not terminal for continuation purposes — the model turn + * (and the client-visible SSE) is still eligible to be resumed past it. + */ terminal: boolean; + /** The literal `finish_reason` string when present (e.g. "stop", "tool_calls", "length", + * "content_filter"), or `null` if none was seen. `terminal` alone is not precise enough + * to gate the reasoning-only-stop continuation — it must fire on `"stop"` only. */ + finishReason: string | null; /** True if at least one OpenAI-shaped `choices[].delta` was parsed (format gate). */ parsedOpenAi: boolean; } @@ -198,11 +221,22 @@ export interface OpenAiSseScan { */ export function scanOpenAiSseText(sse: string): OpenAiSseScan { let text = ""; + let reasoningText = ""; let sawToolCall = false; + let toolCallFinished = false; let terminal = false; + let finishReason: string | null = null; let parsedOpenAi = false; if (typeof sse !== "string" || sse.length === 0) { - return { text, sawToolCall, terminal, parsedOpenAi }; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight: false, + terminal, + finishReason, + parsedOpenAi, + }; } for (const line of sse.split("\n")) { const trimmed = line.trimStart(); @@ -227,14 +261,33 @@ export function scanOpenAiSseText(sse: string): OpenAiSseScan { parsedOpenAi = true; const content = (delta as { content?: unknown }).content; if (typeof content === "string") text += content; + const reasoning = (delta as { reasoning_content?: unknown }).reasoning_content; + if (typeof reasoning === "string") reasoningText += reasoning; const toolCalls = (delta as { tool_calls?: unknown }).tool_calls; if (Array.isArray(toolCalls) && toolCalls.length > 0) sawToolCall = true; } - const finishReason = (choice as { finish_reason?: unknown })?.finish_reason; - if (finishReason != null) terminal = true; + const rawFinishReason = (choice as { finish_reason?: unknown })?.finish_reason; + if (rawFinishReason === "tool_calls") { + // Ends this one choice, but the overall stream/turn stays continuable — + // never counts as the general terminal marker (see OpenAiSseScan.terminal). + toolCallFinished = true; + finishReason = "tool_calls"; + } else if (rawFinishReason != null) { + terminal = true; + if (typeof rawFinishReason === "string") finishReason = rawFinishReason; + } } } - return { text, sawToolCall, terminal, parsedOpenAi }; + const sawToolCallInFlight = sawToolCall && !toolCallFinished; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight, + terminal, + finishReason, + parsedOpenAi, + }; } export interface ContinuableBody { @@ -245,8 +298,10 @@ export interface ContinuableBody { /** * Build a re-request body that continues from `assistantSoFar` by appending it as an - * assistant turn. Returns null when the body has no `messages` array or the partial text - * is empty (nothing to continue from). Does not mutate the original. + * assistant turn. When `assistantSoFar` is empty (nothing usable was emitted yet — e.g. a + * clean stop that only produced reasoning), the messages are re-sent unchanged instead of + * appending an empty assistant turn: this simply re-asks for a real answer. Returns null + * only when the body has no `messages` array at all (nothing to continue from). */ export function makeContinuationBody( body: ContinuableBody, @@ -254,10 +309,13 @@ export function makeContinuationBody( ): (ContinuableBody & { messages: unknown[] }) | null { if (!body || typeof body !== "object") return null; if (!Array.isArray(body.messages) || body.messages.length === 0) return null; - if (typeof assistantSoFar !== "string" || assistantSoFar.length === 0) return null; + if (typeof assistantSoFar !== "string") return null; return { ...body, - messages: [...body.messages, { role: "assistant", content: assistantSoFar }], + messages: + assistantSoFar.length > 0 + ? [...body.messages, { role: "assistant", content: assistantSoFar }] + : [...body.messages], stream: true, }; } @@ -368,8 +426,13 @@ export function createRecoverableStream( let continuations = 0; let emittedTail = ""; // raw SSE not yet scanned (awaiting an event boundary) let emittedText = ""; // assistant text already delivered to the client + let emittedReasoningText = ""; // reasoning trace already delivered (never shown to the client, + // tracked only to distinguish "a real empty turn" from "the whole + // answer stayed in the reasoning channel") + let emittedFinishReason: string | null = null; // literal finish_reason last seen, if any let emittedTerminal = false; - let emittedToolCall = false; + let emittedToolCallInFlight = false; + let emittedSawToolCall = false; // any tool_call delta seen, complete or not let emittedParsedOpenAi = false; // Enqueue to the client and, when continuation is enabled, fold the chunk into the @@ -387,8 +450,11 @@ export function createRecoverableStream( emittedTail = emittedTail.slice(boundary + 2); const scan = scanOpenAiSseText(complete); emittedText += scan.text; + emittedReasoningText += scan.reasoningText; + if (scan.finishReason !== null) emittedFinishReason = scan.finishReason; if (scan.terminal) emittedTerminal = true; - if (scan.sawToolCall) emittedToolCall = true; + if (scan.sawToolCallInFlight) emittedToolCallInFlight = true; + if (scan.sawToolCall) emittedSawToolCall = true; if (scan.parsedOpenAi) emittedParsedOpenAi = true; }; @@ -396,15 +462,42 @@ export function createRecoverableStream( for (const chunk of holdback.flush()) emit(controller, chunk); }; - // A post-commit truncation is continuable only for a plain-text OpenAI-compatible - // stream that has not finished and has no tool call in flight. + // A post-commit truncation is continuable for a plain-text OpenAI-compatible stream that + // has no tool call in flight, AND either: + // - has not finished yet (the original #4131 truncation case), or + // - finished with a literal finish_reason of "stop" but delivered nothing usable while a + // non-empty reasoning trace shows the provider spent its whole turn "thinking" and never + // turned that into an answer (some providers put the entire response in + // reasoning_content and leave content empty). Gated on the LITERAL "stop" value, not the + // generic `terminal` flag — `terminal` also covers "length"/"content_filter"/a bare + // [DONE], which are out of scope for this specific recovery. + // + // Known consequence of the hallucinatedEmptyStop path (flagged in cross-review, accepted as + // inherent to tryContinue's existing design, not new to this fix): the original upstream's + // `finish_reason:"stop"` chunk was already forwarded to the client via `emit()`'s unconditional + // `controller.enqueue(chunk)` (streamRecovery.ts:381) BEFORE this scan ever runs — that is how + // `emittedFinishReason`/`emittedTerminal` get set in the first place. So the client sees an + // empty "stop" marker from the original turn, then — once the continuation succeeds — the real + // answer plus a SECOND `emitCleanTerminal` from `tryContinue`. This mirrors what already + // happens for the pre-existing truncation-continuation case (a truncated stream can likewise + // have partially delivered SSE framing before `tryContinue` appends more); it is not a new + // double-close of the underlying `ReadableStream` (`controller.close()` runs exactly once, + // after `tryContinue` returns). An SSE client that treats a bare `finish_reason:"stop"` as an + // unconditional end-of-turn (rather than waiting for `[DONE]`) may need updating separately — + // out of scope for this fix, which targets the observed opencode/OmniRoute pairing where the + // client kept the connection open. + const hallucinatedEmptyStop = () => + emittedFinishReason === "stop" && + !emittedSawToolCall && + emittedText.length === 0 && + emittedReasoningText.length > 0; + const canContinue = () => continueEnabled && continuations < maxContinuations && emittedParsedOpenAi && - !emittedToolCall && - !emittedTerminal && - emittedText.length > 0; + !emittedToolCallInFlight && + (emittedText.length > 0 ? !emittedTerminal : hallucinatedEmptyStop()); const emitCleanTerminal = (controller: ReadableStreamDefaultController) => { controller.enqueue( @@ -448,7 +541,24 @@ export function createRecoverableStream( } const scan = scanOpenAiSseText(raw); - const suffix = trimContinuationOverlap(emittedText, scan.text); + // A continuation whose overlap with what was already emitted falls below the documented + // threshold is treated as a suspected restart rather than a real resume — see + // STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS for the full trade-off rationale. This + // is a heuristic, not a proof: it deliberately trades some false-positive rejections of + // legitimate low-overlap continuations against never silently gluing two unrelated + // fragments into one corrupted message. + const overlapResult = trimContinuationOverlap(emittedText, scan.text); + const overlapChars = scan.text.length - overlapResult.length; + const isSuspectedRestart = + emittedText.length > 0 && + scan.text.length > 0 && + overlapChars < STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS; + if (isSuspectedRestart) { + if (await tryContinue(controller)) return true; + emitCleanTerminal(controller); + return true; + } + const suffix = overlapResult; if (suffix) { emit( controller, @@ -505,9 +615,11 @@ export function createRecoverableStream( const { done, value } = result; if (done) { if (holdback.committed) { - // Graceful end after commit: if it lacks a terminal marker it is a silent - // truncation — try to continue; otherwise (clean finish) just close. - if (!emittedTerminal && (await tryContinue(controller))) { + // Graceful end after commit: try a mid-stream continuation whenever canContinue() + // says the stream is worth continuing (silent truncation, or a clean-but-empty + // reasoning-only stop) — canContinue() is the single source of truth here, same as + // the read-error branch above. + if (await tryContinue(controller)) { runFinalize(); controller.close(); return; diff --git a/open-sse/services/tokenRefresh.ts b/open-sse/services/tokenRefresh.ts index 2ca3c5df62b..893496846de 100755 --- a/open-sse/services/tokenRefresh.ts +++ b/open-sse/services/tokenRefresh.ts @@ -48,6 +48,7 @@ import { refreshGoogleToken } from "./tokenRefresh/providers/google.ts"; import { ensureAntigravityProjectAssigned } from "./antigravityProjectBootstrap.ts"; import { persistDiscoveredAntigravityProjectId } from "./antigravityProjectPersist.ts"; import { refreshCodexToken } from "./tokenRefresh/providers/codex.ts"; +import { refreshCursorToken } from "./tokenRefresh/providers/cursor.ts"; import { refreshOpenferenceToken } from "./tokenRefresh/providers/openference.ts"; import { refreshKiroToken } from "./tokenRefresh/providers/kiro.ts"; import { refreshQoderToken } from "./tokenRefresh/providers/qoder.ts"; @@ -62,6 +63,7 @@ export { refreshClaudeOAuthToken, refreshGoogleToken, refreshCodexToken, + refreshCursorToken, refreshOpenferenceToken, refreshKiroToken, refreshQoderToken, @@ -382,6 +384,12 @@ async function _getAccessTokenInternal(provider, credentials, log, proxyConfig: case "codex": return await refreshCodexToken(credentials.refreshToken, log, proxyConfig); + case "cursor": + if (!credentials.refreshToken) { + return { error: "unrecoverable_refresh_error", code: "no_refresh_token" }; + } + return await refreshCursorToken(credentials.refreshToken, log, proxyConfig); + case "openference": return await refreshOpenferenceToken(credentials.refreshToken, log, proxyConfig); @@ -453,6 +461,7 @@ export function supportsTokenRefresh(provider) { // testStatus="expired" / errorCode="no_refresh_token". "gitlab-duo", "codebuddy-cn", + "cursor", ]); if (explicitlySupported.has(provider)) return true; const config = PROVIDERS[provider]; diff --git a/open-sse/services/tokenRefresh/providers/cursor.ts b/open-sse/services/tokenRefresh/providers/cursor.ts new file mode 100644 index 00000000000..69a146b07ce --- /dev/null +++ b/open-sse/services/tokenRefresh/providers/cursor.ts @@ -0,0 +1,115 @@ +/** + * Cursor OAuth token refresh via api2.cursor.sh/auth/exchange_user_api_key. + * OpenCodex-compatible (Bearer refresh token, JSON body `{}`). + * Self-contained in open-sse (no import from src/). + */ + +const CURSOR_REFRESH_URL = "https://api2.cursor.sh/auth/exchange_user_api_key"; +const REFRESH_TIMEOUT_MS = 15_000; +const REFRESH_ATTEMPTS = 3; +const REFRESH_RETRY_BASE_MS = 300; +const EXPIRY_SKEW_MS = 5 * 60 * 1000; +const FALLBACK_TTL_MS = 60 * 60 * 1000; + +function isRetryableRefreshStatus(status: number): boolean { + return status === 429 || status === 500 || status === 502 || status === 503 || status === 504; +} + +function refreshRetryDelayMs(attempt: number, baseMs: number): number { + const exp = baseMs * 2 ** attempt; + return Math.floor(exp * (0.8 + Math.random() * 0.4)); +} + +function decodeExpMs(token: string): number { + try { + const parts = token.split("."); + if (parts.length !== 3) return Date.now() + FALLBACK_TTL_MS; + const payload = JSON.parse(Buffer.from(parts[1], "base64url").toString("utf-8")) as { + exp?: unknown; + }; + if (typeof payload.exp === "number") return payload.exp * 1000 - EXPIRY_SKEW_MS; + } catch { + /* ignore */ + } + return Date.now() + FALLBACK_TTL_MS; +} + +export type RefreshCursorTokenOptions = { + retryBaseMs?: number; + attempts?: number; +}; + +/** + * @returns {{ accessToken, refreshToken, expiresAt } | { error, code } | null} + */ +export async function refreshCursorToken( + refreshToken: string, + log?: { error?: (...args: unknown[]) => void; info?: (...args: unknown[]) => void }, + _proxyConfig: unknown = null, + options: RefreshCursorTokenOptions = {} +) { + if (!refreshToken) { + return { error: "unrecoverable_refresh_error", code: "no_refresh_token" }; + } + + const attempts = options.attempts ?? REFRESH_ATTEMPTS; + const retryBaseMs = options.retryBaseMs ?? REFRESH_RETRY_BASE_MS; + let lastError: unknown; + + for (let attempt = 0; attempt < attempts; attempt++) { + let response: Response; + try { + response = await fetch(CURSOR_REFRESH_URL, { + method: "POST", + headers: { + Authorization: `Bearer ${refreshToken}`, + "Content-Type": "application/json", + }, + body: "{}", + signal: AbortSignal.timeout(REFRESH_TIMEOUT_MS), + }); + } catch (err) { + lastError = err; + if (attempt === attempts - 1) break; + await new Promise((r) => setTimeout(r, refreshRetryDelayMs(attempt, retryBaseMs))); + continue; + } + + if (response.ok) { + const data = (await response.json()) as { accessToken?: string; refreshToken?: string }; + if (!data.accessToken) { + log?.error?.("TOKEN_REFRESH", "Cursor refresh response missing access token"); + return null; + } + const nextRefresh = data.refreshToken || refreshToken; + log?.info?.("TOKEN_REFRESH", "Successfully refreshed Cursor token"); + return { + accessToken: data.accessToken, + refreshToken: nextRefresh, + expiresAt: new Date(decodeExpMs(data.accessToken)).toISOString(), + }; + } + + if (response.status === 401 || response.status === 403) { + log?.error?.("TOKEN_REFRESH", "Cursor refresh rejected — re-authentication required", { + status: response.status, + }); + return { error: "unrecoverable_refresh_error", code: "unauthorized" }; + } + + if (!isRetryableRefreshStatus(response.status) || attempt === attempts - 1) { + log?.error?.("TOKEN_REFRESH", "Failed to refresh Cursor token", { status: response.status }); + return null; + } + + lastError = new Error(`Cursor token refresh failed: ${response.status}`); + await response.body?.cancel().catch(() => {}); + await new Promise((r) => setTimeout(r, refreshRetryDelayMs(attempt, retryBaseMs))); + } + + log?.error?.( + "TOKEN_REFRESH", + lastError instanceof Error ? lastError.message : "Cursor token refresh failed" + ); + return null; +} diff --git a/open-sse/services/usage/cursor.ts b/open-sse/services/usage/cursor.ts index 67acc09a6a3..a0d15ee0b1c 100644 --- a/open-sse/services/usage/cursor.ts +++ b/open-sse/services/usage/cursor.ts @@ -1,21 +1,24 @@ /** * usage/cursor.ts — Cursor (Pro) usage fetcher + JWT/config helpers. * - * Extracted from services/usage.ts (god-file decomposition): the Cursor family — the - * dashboard usage-API config, the WorkOS JWT `sub` decoder, and the getCursorUsage fetcher - * that probes the cursor.com/dashboard/spending endpoint. Depends only on the sibling - * scalar/quota leaves — no host coupling — so it lives as a co-located provider leaf. - * usage.ts imports getCursorUsage (dispatcher). Behavior-preserving move. + * Prefer Bearer APIs on api2.cursor.sh (works with deep-control PKCE JWTs). + * Fall back to the cookie-based cursor.com dashboard endpoint for IDE-imported + * WorkOS sessions. OpenCodex-compatible chain: + * GetCurrentPeriodUsage → /api/usage/summary → /auth/usage → cookie dashboard. */ import { toRecord, toNumber, clampPercentage } from "./scalars.ts"; import { type UsageQuota, parseResetTime } from "./quota.ts"; -// Cursor dashboard usage API config -// The endpoint that powers https://cursor.com/dashboard/spending. Validates the WorkOS -// session via the WorkosCursorSessionToken cookie (format: `${userId}::${jwt}`) and -// rejects requests without a matching Origin/Referer (Invalid origin for state-changing request). -const CURSOR_USAGE_CONFIG = { +const REQUEST_TIMEOUT_MS = 12_000; + +const CURSOR_API2 = "https://api2.cursor.sh"; +const CURSOR_PERIOD_USAGE_URL = `${CURSOR_API2}/aiserver.v1.DashboardService/GetCurrentPeriodUsage`; +const CURSOR_USAGE_SUMMARY_URL = `${CURSOR_API2}/api/usage/summary`; +const CURSOR_AUTH_USAGE_URL = `${CURSOR_API2}/auth/usage`; + +/** Legacy IDE/session cookie path (last resort). */ +const CURSOR_COOKIE_USAGE_CONFIG = { usageUrl: "https://cursor.com/api/dashboard/get-current-period-usage", origin: "https://cursor.com", referer: "https://cursor.com/dashboard/spending", @@ -23,11 +26,19 @@ const CURSOR_USAGE_CONFIG = { "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", }; +const REAUTH_HINT = "Use Cursor Login (PKCE) or re-import the connection from Cursor IDE."; + +export type CursorUsageResult = { + plan?: string; + quotas?: Record; + message?: string; +}; + /** * Decode the `sub` claim of a Cursor JWT (the WorkOS user id). * Returns null if the token is not a parseable JWT. */ -function decodeCursorJwtSub(token: string): string | null { +export function decodeCursorJwtSub(token: string): string | null { if (!token || typeof token !== "string") return null; const parts = token.split("."); if (parts.length !== 3) return null; @@ -42,120 +53,306 @@ function decodeCursorJwtSub(token: string): string | null { } } -/** - * Cursor Pro Plan Usage - * Fetches current-billing-cycle spend from the cursor.com dashboard API and exposes three - * windows that mirror the cursor.com/dashboard/spending UI: Total / Auto + Composer / API. - */ -export async function getCursorUsage(accessToken: string, providerSpecificData?: unknown) { - if (!accessToken) { - return { message: "Cursor access token missing. Re-import the connection from Cursor IDE." }; +function bearerHeaders(accessToken: string): Record { + return { + Accept: "application/json", + Authorization: `Bearer ${accessToken}`, + "User-Agent": "omniroute-cursor-quota", + }; +} + +function toDollars(cents: number): number { + return Math.round(cents) / 100; +} + +function buildPlanUsageQuotas( + planUsage: Record, + billingCycleEnd: unknown +): Record | null { + const limitCents = Math.max( + 0, + toNumber(planUsage.limit ?? planUsage.limitCents ?? planUsage.totalLimitCents, 0) + ); + const includedSpendRaw = toNumber( + planUsage.includedSpend ?? planUsage.usedCents ?? planUsage.used, + NaN + ); + const totalSpendCents = Number.isFinite(includedSpendRaw) + ? Math.max(0, includedSpendRaw) + : Math.max(0, toNumber(planUsage.totalSpend, 0)); + + const rawTotalPct = toNumber(planUsage.totalPercentUsed ?? planUsage.percentUsed, NaN); + let totalPercentUsed: number; + if (Number.isFinite(rawTotalPct)) { + totalPercentUsed = clampPercentage(rawTotalPct); + } else if (limitCents > 0) { + totalPercentUsed = clampPercentage((totalSpendCents / limitCents) * 100); + } else { + return null; } - const storedUserId = (() => { - const raw = toRecord(providerSpecificData).userId; - return typeof raw === "string" && raw.length > 0 ? raw : null; - })(); - const userId = storedUserId || decodeCursorJwtSub(accessToken); + const autoPercentUsed = clampPercentage(toNumber(planUsage.autoPercentUsed, 0)); + const apiPercentUsed = clampPercentage(toNumber(planUsage.apiPercentUsed, 0)); + const effectiveLimitCents = limitCents > 0 ? limitCents : 100; - if (!userId) { + const billingCycleEndMs = toNumber(billingCycleEnd, 0); + const resetAt = billingCycleEndMs > 0 ? parseResetTime(billingCycleEndMs) : null; + const limitDollars = toDollars(effectiveLimitCents); + + const buildWindow = (percentUsed: number, usedCentsOverride?: number): UsageQuota => { + const usedCents = + typeof usedCentsOverride === "number" + ? usedCentsOverride + : Math.round((effectiveLimitCents * percentUsed) / 100); + const clampedUsed = Math.min(usedCents, effectiveLimitCents); return { - message: "Cursor token missing user id. Re-import the connection from Cursor IDE.", + used: toDollars(clampedUsed), + total: limitDollars, + remaining: toDollars(Math.max(effectiveLimitCents - clampedUsed, 0)), + remainingPercentage: clampPercentage(100 - percentUsed), + resetAt, + unlimited: false, }; + }; + + return { + Total: buildWindow(totalPercentUsed, limitCents > 0 ? totalSpendCents : undefined), + "Auto + Composer": buildWindow(autoPercentUsed), + API: buildWindow(apiPercentUsed), + }; +} + +async function fetchJson( + url: string, + init: RequestInit +): Promise<{ ok: true; data: Record } | { ok: false }> { + try { + const response = await fetch(url, { + ...init, + signal: init.signal ?? AbortSignal.timeout(REQUEST_TIMEOUT_MS), + }); + if (!response.ok) return { ok: false }; + const data = toRecord(await response.json().catch(() => null)); + if (Object.keys(data).length === 0) return { ok: false }; + return { ok: true, data }; + } catch { + return { ok: false }; + } +} + +function tryPeriodUsage(data: Record): CursorUsageResult | null { + const planUsage = toRecord(data.planUsage); + if (Object.keys(planUsage).length === 0) return null; + const quotas = buildPlanUsageQuotas(planUsage, data.billingCycleEnd ?? planUsage.billingCycleEnd); + if (!quotas) return null; + return { plan: "Cursor Pro", quotas }; +} + +function tryUsageSummary(data: Record): CursorUsageResult | null { + const individual = toRecord(data.individualUsage); + const plan = toRecord(individual.plan); + if (Object.keys(plan).length === 0) return null; + const used = toNumber(plan.used, NaN); + const limit = toNumber(plan.limit, NaN); + const percent = clampPercentage( + toNumber( + plan.totalPercentUsed, + Number.isFinite(used) && Number.isFinite(limit) && limit > 0 ? (used / limit) * 100 : NaN + ) + ); + if ( + !Number.isFinite(toNumber(plan.totalPercentUsed, NaN)) && + !(Number.isFinite(used) && limit > 0) + ) { + return null; } + const quotas = buildPlanUsageQuotas( + { + limit: Number.isFinite(limit) ? limit : 100, + totalSpend: Number.isFinite(used) ? used : Math.round(percent), + totalPercentUsed: percent, + autoPercentUsed: percent, + apiPercentUsed: 0, + }, + data.billingCycleEnd + ); + if (!quotas) return null; + return { plan: "Cursor Pro", quotas }; +} +function tryAuthUsage(data: Record): CursorUsageResult | null { + let used: number | undefined; + let limit: number | undefined; + const gpt4 = toRecord(data["gpt-4"]); + if (Object.keys(gpt4).length > 0) { + used = toNumber(gpt4.numRequests ?? gpt4.used, NaN); + limit = toNumber(gpt4.maxRequestUsage ?? gpt4.limit ?? gpt4.maxRequests, NaN); + } + if (!Number.isFinite(used) || !Number.isFinite(limit) || (limit as number) <= 0) { + for (const [key, value] of Object.entries(data)) { + if (key === "startOfMonth" || key === "billingCycleStart") continue; + const bucket = toRecord(value); + if (Object.keys(bucket).length === 0) continue; + const bucketUsed = toNumber(bucket.numRequests ?? bucket.used, NaN); + const bucketLimit = toNumber( + bucket.maxRequestUsage ?? bucket.limit ?? bucket.maxRequests, + NaN + ); + if (Number.isFinite(bucketUsed) && Number.isFinite(bucketLimit) && bucketLimit > 0) { + used = bucketUsed; + limit = bucketLimit; + break; + } + } + } + if (!Number.isFinite(used) || !Number.isFinite(limit) || (limit as number) <= 0) return null; + const percent = clampPercentage(((used as number) / (limit as number)) * 100); + const startOfMonth = parseResetTime(data.startOfMonth ?? data.billingCycleStart); + let monthlyResetAt: string | null = null; + if (startOfMonth) { + const start = new Date(startOfMonth); + monthlyResetAt = new Date( + Date.UTC(start.getUTCFullYear(), start.getUTCMonth() + 1, start.getUTCDate()) + ).toISOString(); + } + const limitNum = limit as number; + const usedNum = used as number; + return { + plan: "Cursor Pro", + quotas: { + Total: { + used: usedNum, + total: limitNum, + remaining: Math.max(limitNum - usedNum, 0), + remainingPercentage: clampPercentage(100 - percent), + resetAt: monthlyResetAt, + unlimited: false, + }, + }, + }; +} + +async function fetchCookieDashboardUsage( + accessToken: string, + userId: string +): Promise { try { - const response = await fetch(CURSOR_USAGE_CONFIG.usageUrl, { + const response = await fetch(CURSOR_COOKIE_USAGE_CONFIG.usageUrl, { method: "POST", redirect: "manual", headers: { Cookie: `WorkosCursorSessionToken=${userId}::${accessToken}`, - Origin: CURSOR_USAGE_CONFIG.origin, - Referer: CURSOR_USAGE_CONFIG.referer, + Origin: CURSOR_COOKIE_USAGE_CONFIG.origin, + Referer: CURSOR_COOKIE_USAGE_CONFIG.referer, "Content-Type": "application/json", Accept: "application/json", - "User-Agent": CURSOR_USAGE_CONFIG.userAgent, + "User-Agent": CURSOR_COOKIE_USAGE_CONFIG.userAgent, }, body: "{}", + signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS), }); - // 3xx redirect to WorkOS authkit means the session cookie was rejected. if (response.status >= 300 && response.status < 400) { return { plan: "Cursor", - message: "Cursor session expired. Re-import the token from Cursor IDE.", + message: `Cursor session expired. ${REAUTH_HINT}`, }; } if (!response.ok) { - const errorText = (await response.text()).slice(0, 200); if (response.status === 401 || response.status === 403) { return { plan: "Cursor", - message: "Cursor session unauthorized. Re-import the token from Cursor IDE.", + message: `Cursor session unauthorized. ${REAUTH_HINT}`, }; } return { plan: "Cursor", - message: `Cursor usage endpoint error (${response.status}): ${errorText}`, + message: `Cursor usage endpoint error (${response.status}). ${REAUTH_HINT}`, }; } const data = toRecord(await response.json()); const planUsage = toRecord(data.planUsage); - if (Object.keys(planUsage).length === 0) { return { plan: "Cursor", message: "Cursor connected. No active plan usage returned.", }; } - - const limitCents = Math.max(0, toNumber(planUsage.limit, 0)); - const totalSpendCents = Math.max(0, toNumber(planUsage.totalSpend, 0)); - const autoPercentUsed = clampPercentage(toNumber(planUsage.autoPercentUsed, 0)); - const apiPercentUsed = clampPercentage(toNumber(planUsage.apiPercentUsed, 0)); - const totalPercentUsed = clampPercentage(toNumber(planUsage.totalPercentUsed, 0)); - - // billingCycleEnd is a numeric-string in ms; coerce so parseResetTime sees a number. - const billingCycleEndMs = toNumber(data.billingCycleEnd, 0); - const resetAt = billingCycleEndMs > 0 ? parseResetTime(billingCycleEndMs) : null; - - // Convert cents → dollars rounded to 2 decimal places. - const toDollars = (cents: number) => Math.round(cents) / 100; - - const limitDollars = toDollars(limitCents); - const buildWindow = (percentUsed: number, usedCentsOverride?: number): UsageQuota => { - const usedCents = - typeof usedCentsOverride === "number" - ? usedCentsOverride - : Math.round((limitCents * percentUsed) / 100); - const used = toDollars(Math.min(usedCents, limitCents)); - const remaining = toDollars(Math.max(limitCents - Math.min(usedCents, limitCents), 0)); + const quotas = buildPlanUsageQuotas(planUsage, data.billingCycleEnd); + if (!quotas) { return { - used, - total: limitDollars, - remaining, - remainingPercentage: clampPercentage(100 - percentUsed), - resetAt, - unlimited: false, + plan: "Cursor", + message: "Cursor connected. No active plan usage returned.", }; + } + return { plan: "Cursor Pro", quotas }; + } catch (error) { + return { + plan: "Cursor", + message: `Cursor connected. Unable to fetch usage: ${(error as Error).message}`, }; + } +} - const quotas: Record = { - Total: buildWindow(totalPercentUsed, totalSpendCents), - "Auto + Composer": buildWindow(autoPercentUsed), - API: buildWindow(apiPercentUsed), - }; +/** + * Cursor Pro Plan Usage — Bearer APIs first (PKCE), cookie dashboard last (IDE import). + */ +export async function getCursorUsage( + accessToken: string, + providerSpecificData?: unknown +): Promise { + if (!accessToken) { + return { message: `Cursor access token missing. ${REAUTH_HINT}` }; + } - return { - plan: "Cursor Pro", - quotas, - }; - } catch (error) { + const auth = bearerHeaders(accessToken); + + const period = await fetchJson(CURSOR_PERIOD_USAGE_URL, { + method: "POST", + headers: { + ...auth, + "Content-Type": "application/json", + "Connect-Protocol-Version": "1", + }, + body: "{}", + }); + if (period.ok) { + const mapped = tryPeriodUsage(period.data); + if (mapped) return mapped; + } + + const summary = await fetchJson(CURSOR_USAGE_SUMMARY_URL, { + method: "GET", + headers: auth, + }); + if (summary.ok) { + const mapped = tryUsageSummary(summary.data); + if (mapped) return mapped; + } + + const authUsage = await fetchJson(CURSOR_AUTH_USAGE_URL, { + method: "GET", + headers: auth, + }); + if (authUsage.ok) { + const mapped = tryAuthUsage(authUsage.data); + if (mapped) return mapped; + } + + const storedUserId = (() => { + const raw = toRecord(providerSpecificData).userId; + return typeof raw === "string" && raw.length > 0 ? raw : null; + })(); + const userId = storedUserId || decodeCursorJwtSub(accessToken); + + if (!userId) { return { plan: "Cursor", - message: `Cursor connected. Unable to fetch usage: ${(error as Error).message}`, + message: `Cursor usage unavailable via API and token has no user id for cookie fallback. ${REAUTH_HINT}`, }; } + + return fetchCookieDashboardUsage(accessToken, userId); } diff --git a/open-sse/translator/request/openai-responses/toResponses.ts b/open-sse/translator/request/openai-responses/toResponses.ts index bee5efab1a2..988835edeac 100644 --- a/open-sse/translator/request/openai-responses/toResponses.ts +++ b/open-sse/translator/request/openai-responses/toResponses.ts @@ -201,6 +201,14 @@ export function openaiToOpenAIResponsesRequest( input.push({ type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], + // Strict Responses-API upstreams (e.g. opencode/zen) require `summary` + // on every `input[]` item of type "reasoning", plaintext or opaque — + // omitting it rejects the request with `input[N] missing required + // field summary`. This item is always freshly built from a chat + // client's plaintext reasoning, so there is no source summary to + // preserve; default to an empty array like the replay sanitizer does + // for opaque items in reasoningInputPolicy.ts (#11108). + summary: [], }); } diff --git a/open-sse/translator/response/openai-responses.ts b/open-sse/translator/response/openai-responses.ts index 0c59fac4fba..01ad55d72f2 100644 --- a/open-sse/translator/response/openai-responses.ts +++ b/open-sse/translator/response/openai-responses.ts @@ -866,21 +866,25 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { function openaiResponsesToOpenAIResponseStream(chunk, state) { if (!chunk) { - if ( - state.currentToolCallNeedsNormalization && - state.currentToolCallArgsBuffer && - state.currentToolCallName - ) { - const toolSchema = state.toolSchemas?.get(state.currentToolCallName); - const argsToEmit = stripEmptyOptionalToolArgs( - state.currentToolCallArgsBuffer, - state.currentToolCallName, - toolSchema - ); - const argsStr = - typeof argsToEmit === "string" ? argsToEmit : JSON.stringify(argsToEmit ?? {}); - state.currentToolCallArgsBuffer = ""; - state.currentToolCallNeedsNormalization = false; + // Iterate every still-open call needing schema-aware normalization, not just a + // single one — multiple parallel calls can each be pending here if the stream + // ends before their output_item.done arrives. + const pendingNormalized: Array<{ index: number; argsStr: string }> = []; + if (state.toolCallByCallId instanceof Map) { + for (const entry of state.toolCallByCallId.values()) { + if (entry.needsNormalization && entry.argsBuffer) { + const toolSchema = state.toolSchemas?.get(entry.name); + const argsToEmit = stripEmptyOptionalToolArgs(entry.argsBuffer, entry.name, toolSchema); + pendingNormalized.push({ + index: entry.index, + argsStr: typeof argsToEmit === "string" ? argsToEmit : JSON.stringify(argsToEmit ?? {}), + }); + entry.argsBuffer = ""; + entry.needsNormalization = false; + } + } + } + if (pendingNormalized.length > 0) { state.finishReasonSent = true; state.finishReason = "tool_calls"; const common = { @@ -889,24 +893,21 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { created: state.created, model: state.model || "gpt-4", }; - return [ - { - ...common, - choices: [ - { - index: 0, - delta: { - tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsStr } }], - }, - finish_reason: null, - }, - ], - }, - { - ...common, - choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }], - }, - ]; + const chunks: Record[] = pendingNormalized.map(({ index, argsStr }) => ({ + ...common, + choices: [ + { + index: 0, + delta: { tool_calls: [{ index, function: { arguments: argsStr } }] }, + finish_reason: null, + }, + ], + })); + chunks.push({ + ...common, + choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }], + }); + return chunks; } // Flush: send final chunk with finish_reason if (!state.finishReasonSent && state.started) { @@ -952,7 +953,23 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { state.chatId = `chatcmpl-${Date.now()}`; state.created = Math.floor(Date.now() / 1000); state.toolCallIndex = 0; + // Kept for computeFinishReason (synthesizeCompletedToolCalls.ts) compatibility — + // that snapshot path mutates it directly and expects it to exist. In a turn with + // multiple parallel calls this only ever reflects the LAST one opened/closed, so + // it must never be used to identify a specific call — only as the "is at least + // one tool call in flight this turn" signal computeFinishReason needs, which + // toolCallIndex > 0 already covers on its own once any call has been added. state.currentToolCallId = null; + // Per-call state keyed by call_id (replaces the old singular + // currentToolCallId/ArgsBuffer/Name/NeedsNormalization/Deferred fields, which + // assumed only one function_call could ever be in flight at a time). + state.toolCallByCallId = new Map(); + // response.function_call_arguments.delta carries `item_id`/`output_index`, not + // `call_id` — resolve either one back to the call_id key used by + // toolCallByCallId (two independent reverse maps, since some upstreams omit + // item_id on delta events but still send output_index). + state.toolCallItemToCallId = new Map(); + state.toolCallOutputIndexToCallId = new Map(); } // Text content delta @@ -983,22 +1000,48 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { // Function call started if (eventType === "response.output_item.added" && data.item?.type === "function_call") { const item = data.item; - state.currentToolCallId = item.call_id || fallbackToolCallId(); - state.currentToolCallArgsBuffer = ""; // reset per-call arg buffer - state.currentToolCallDeferred = false; + const callId = item.call_id || fallbackToolCallId(); + // Kept for computeFinishReason (synthesizeCompletedToolCalls.ts) compatibility. + state.currentToolCallId = callId; + + const toolName = normalizeToolName(item.name); + // Assign this call's index NOW, at .added, not at .done — two calls opened before + // either closes (a genuine parallel dispatch) must never share an index. Deferred + // (still-nameless) calls are the one exception: they don't claim an index until + // .done resolves a real name, so a call that never gets one never burns a slot + // another call could have used. + let index: number | null = null; + if (toolName) { + index = state.toolCallIndex ?? 0; + state.toolCallIndex = index + 1; + } + + if (!(state.toolCallByCallId instanceof Map)) state.toolCallByCallId = new Map(); + state.toolCallByCallId.set(callId, { + index, + name: toolName, + argsBuffer: "", + deferred: !toolName, + needsNormalization: toolName === "Agent", + }); + if (!(state.toolCallItemToCallId instanceof Map)) state.toolCallItemToCallId = new Map(); + if (item.id) state.toolCallItemToCallId.set(item.id, callId); + // `output_index` is a top-level field on every Responses API streamed event + // (response.output_item.added/.done AND function_call_arguments.delta alike) — + // an identifier independent of item_id, for upstreams that omit item_id on delta + // events. + if (!(state.toolCallOutputIndexToCallId instanceof Map)) { + state.toolCallOutputIndexToCallId = new Map(); + } + if (data.output_index != null) state.toolCallOutputIndexToCallId.set(data.output_index, callId); // Track this call_id so response.completed doesn't synthesize a duplicate if (!state.toolCallIdsSeen) state.toolCallIdsSeen = new Set(); - if (state.currentToolCallId) state.toolCallIdsSeen.add(state.currentToolCallId); + state.toolCallIdsSeen.add(callId); - const toolName = normalizeToolName(item.name); - state.currentToolName = toolName; // track for schema lookup at done time - state.currentToolCallName = toolName; - state.currentToolCallNeedsNormalization = toolName === "Agent"; if (!toolName) { // Some Responses providers briefly emit placeholder/empty tool names. // Defer emission until output_item.done in case the final name is populated there. - state.currentToolCallDeferred = true; return null; } @@ -1013,8 +1056,8 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { delta: { tool_calls: [ { - index: state.toolCallIndex, - id: state.currentToolCallId, + index, + id: callId, type: "function", function: { name: toolName, @@ -1037,11 +1080,26 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { const argsDelta = data.delta || ""; if (!argsDelta) return null; - state.currentToolCallArgsBuffer = (state.currentToolCallArgsBuffer || "") + argsDelta; - if (state.currentToolCallDeferred || state.currentToolCallNeedsNormalization) return null; + // Resolve which in-flight call this delta belongs to. Try item_id first (the + // field the Responses API documents for this event), then output_index (also a + // top-level field on this event, and independent of item_id — covers upstreams + // that omit item_id on delta events but still send output_index). Only once both + // identifying fields are absent/unresolved do we fall back to guessing (the + // single open call, or the most recently opened one as a last resort). + const map = state.toolCallByCallId instanceof Map ? state.toolCallByCallId : null; + let callId = data.item_id ? state.toolCallItemToCallId?.get(data.item_id) : undefined; + if (!callId && data.output_index != null) { + callId = state.toolCallOutputIndexToCallId?.get(data.output_index); + } + if (!callId && map) { + callId = map.size === 1 ? [...map.keys()][0] : state.currentToolCallId; + } + const entry = callId ? map?.get(callId) : undefined; + if (!entry) return null; // #9168: buffer arguments until output_item.done for schema-aware null normalization // Previously emitted raw null values for optional enum fields (e.g. isolation: null). + entry.argsBuffer = (entry.argsBuffer || "") + argsDelta; return null; } @@ -1061,13 +1119,30 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { // carry the complete arguments only in output_item.done (no preceding delta events). if (eventType === "response.output_item.done" && data.item?.type === "function_call") { const item = data.item; - const buffered = state.currentToolCallArgsBuffer || ""; - const currentIndex = state.toolCallIndex; // capture before increment - const callId = item.call_id || state.currentToolCallId || fallbackToolCallId(); + const map = state.toolCallByCallId instanceof Map ? state.toolCallByCallId : null; + let callId = item.call_id; + if (!callId && item.id) callId = state.toolCallItemToCallId?.get(item.id); + if (!callId) callId = state.currentToolCallId || fallbackToolCallId(); + const trackedEntry = callId ? map?.get(callId) : undefined; + // Some upstreams (e.g. Codex) send the complete payload only in output_item.done, + // with no preceding output_item.added at all — there is no tracked entry to read an + // index from. + const entry = trackedEntry || { index: null, argsBuffer: "", deferred: false }; + + const buffered = entry.argsBuffer || ""; const toolName = normalizeToolName(item.name); + + // Claim (and advance) this call's index now if it wasn't assigned at .added — either + // a deferred call whose name has just now resolved, or a Codex-style done-only + // payload that never had an .added at all. A deferred call whose name is STILL empty + // never claims an index (nothing was ever emitted for it either way). + if (entry.index == null && toolName) { + entry.index = state.toolCallIndex ?? 0; + state.toolCallIndex = entry.index + 1; + } + const currentIndex = entry.index; const toolSchema = state.toolSchemas?.get(toolName); const shouldNormalizeArguments = toolName === "Agent"; - state.currentToolCallNeedsNormalization = shouldNormalizeArguments; if (toolName && state.toolCalls instanceof Map) { const completedArguments = @@ -1077,6 +1152,9 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { toolName, toolSchema ); + // Keyed by index, not insertion order — readers that need call order for + // parallel calls closed out of order should sort by this key rather than + // relying on Map iteration order. state.toolCalls.set(currentIndex, { id: callId, index: currentIndex, @@ -1095,17 +1173,17 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { if (!state.toolCallIdsSeen) state.toolCallIdsSeen = new Set(); if (callId) state.toolCallIdsSeen.add(callId); - if (state.currentToolCallDeferred) { - state.currentToolCallDeferred = false; - state.currentToolCallArgsBuffer = ""; - state.currentToolCallId = null; + // This call is fully closed — remove it from the in-flight map (bounds the map + // to genuinely in-flight calls, and keeps the single-open-call fallback in the + // function_call_arguments.delta handler correct for whichever call opens next). + if (map && callId) map.delete(callId); + if (state.currentToolCallId === callId) state.currentToolCallId = null; + if (entry.deferred) { if (!toolName) { return null; } - state.toolCallIndex++; - const terminalArguments = typeof item.arguments === "string" ? item.arguments.length > 0 @@ -1148,12 +1226,7 @@ function openaiResponsesToOpenAIResponseStream(chunk, state) { }; } - state.toolCallIndex++; - state.currentToolCallArgsBuffer = ""; // reset for next tool call - state.currentToolCallId = null; - const needsNormalization = state.currentToolCallNeedsNormalization === true; - state.currentToolCallNeedsNormalization = false; - state.currentToolCallName = ""; + const needsNormalization = shouldNormalizeArguments; // Nullable omission sentinels must be normalized before any argument bytes reach the client. // Other tool calls retain immediate argument streaming. diff --git a/open-sse/utils/cursorAgentCliVersion.ts b/open-sse/utils/cursorAgentCliVersion.ts index 2a65df051b4..91bedd22063 100644 --- a/open-sse/utils/cursorAgentCliVersion.ts +++ b/open-sse/utils/cursorAgentCliVersion.ts @@ -4,10 +4,19 @@ * Wire header: `x-cursor-client-version: cli-${id}` where `id` is a dated * build like `2026.07.08-0c04a8a` (not the IDE `3.x` semver). * - * Resolution: CURSOR_AGENT_CLI_VERSION env → local install detect → pin. + * Resolution: CURSOR_AGENT_CLI_VERSION env → local install detect → + * disk-cached installer scrape (stale-while-revalidate) → pin. */ -import { existsSync, lstatSync, readdirSync, realpathSync } from "node:fs"; +import { + existsSync, + lstatSync, + mkdirSync, + readFileSync, + readdirSync, + realpathSync, + writeFileSync, +} from "node:fs"; import { homedir } from "node:os"; import { join } from "node:path"; @@ -19,9 +28,19 @@ export const CURSOR_AGENT_CLI_VERSION = "2026.07.08-0c04a8a"; const VERSION_ID_RE = /^\d{4}\.\d{2}\.\d{2}-[0-9a-f]+$/; const CACHE_TTL_MS = 60 * 60 * 1000; +const INSTALL_URL = "https://cursor.com/install"; +const REMOTE_TIMEOUT_MS = 5_000; +const VERSION_CACHE_FILE = "cursor-agent-cli-version.json"; let cachedVersion: string | null = null; let cachedAt = 0; +let remoteRefreshInFlight: Promise | null = null; +let remoteRefreshScheduled = false; + +/** Test seam: override fetch for installer scrape. */ +let fetchImpl: typeof fetch = fetch; +/** Test seam: override disk cache directory. */ +let cacheDirOverride: string | null = null; export function isCursorAgentCliVersionId(value: string): boolean { return VERSION_ID_RE.test(value); @@ -43,17 +62,26 @@ export function extractVersionIdFromResolvedPath(resolvedPath: string): string | export function newestVersionInDir(versionsDir: string): string | null { try { if (!existsSync(versionsDir)) return null; - const matches = readdirSync(versionsDir) - .filter((name) => { - if (!isCursorAgentCliVersionId(name)) return false; - try { - return lstatSync(join(versionsDir, name)).isDirectory(); - } catch { - return false; + // Prefer newest mtime (oakimov), break ties with lexicographic id. + let newest: { name: string; mtimeMs: number } | null = null; + for (const name of readdirSync(versionsDir)) { + if (!isCursorAgentCliVersionId(name)) continue; + try { + const st = lstatSync(join(versionsDir, name)); + if (!st.isDirectory()) continue; + const mtimeMs = st.mtimeMs; + if ( + !newest || + mtimeMs > newest.mtimeMs || + (mtimeMs === newest.mtimeMs && name > newest.name) + ) { + newest = { name, mtimeMs }; } - }) - .sort(); - return matches.length > 0 ? matches[matches.length - 1] : null; + } catch { + /* skip vanished entries */ + } + } + return newest?.name ?? null; } catch { return null; } @@ -93,6 +121,80 @@ export function detectCursorAgentCliVersionFromFs(home: string = homedir()): str return newestVersionInDir(versionsDir); } +type DiskVersionCache = { version: string; fetchedAt: number }; + +function resolveCacheDir(): string { + if (cacheDirOverride) return cacheDirOverride; + const dataDir = process.env.DATA_DIR?.trim(); + if (dataDir) return join(dataDir, "cache"); + return join(homedir(), ".omniroute", "cache"); +} + +function versionCachePath(): string { + return join(resolveCacheDir(), VERSION_CACHE_FILE); +} + +export function extractVersionIdFromInstallerScript(script: string): string | null { + const match = script.match(/downloads\.cursor\.com\/lab\/([^/"'\s]+)\//); + if (!match) return null; + const id = match[1]; + return isCursorAgentCliVersionId(id) ? id : null; +} + +function readDiskVersionCache(): DiskVersionCache | null { + try { + const raw = JSON.parse(readFileSync(versionCachePath(), "utf8")) as Record; + if (typeof raw.version !== "string" || !isCursorAgentCliVersionId(raw.version)) return null; + if (typeof raw.fetchedAt !== "number" || !Number.isFinite(raw.fetchedAt)) return null; + return { version: raw.version, fetchedAt: raw.fetchedAt }; + } catch { + return null; + } +} + +function writeDiskVersionCache(cache: DiskVersionCache): void { + try { + const dir = resolveCacheDir(); + mkdirSync(dir, { recursive: true }); + writeFileSync(versionCachePath(), JSON.stringify(cache, null, 2)); + } catch { + // Cache writes are best-effort. + } +} + +async function fetchInstallerVersionId(): Promise { + const response = await fetchImpl(INSTALL_URL, { + signal: AbortSignal.timeout(REMOTE_TIMEOUT_MS), + }); + if (!response.ok) return null; + const text = await response.text(); + return extractVersionIdFromInstallerScript(text); +} + +function scheduleRemoteVersionRefresh(): void { + if (remoteRefreshInFlight || remoteRefreshScheduled) return; + // Defer so sync header resolution never starts network in the same turn. + remoteRefreshScheduled = true; + setTimeout(() => { + remoteRefreshScheduled = false; + if (remoteRefreshInFlight) return; + remoteRefreshInFlight = (async () => { + try { + const id = await fetchInstallerVersionId(); + if (id) writeDiskVersionCache({ version: id, fetchedAt: Date.now() }); + } catch { + // Ignore — pin / stale cache remain valid. + } finally { + remoteRefreshInFlight = null; + } + })(); + }, 0); +} + +/** + * Resolve CLI build id synchronously for request headers. + * Env → local FS → disk cache (refresh in background if stale) → pin. + */ export function getCursorAgentCliVersion(): string { const now = Date.now(); if (cachedVersion && now - cachedAt < CACHE_TTL_MS) { @@ -114,11 +216,57 @@ export function getCursorAgentCliVersion(): string { return cachedVersion; } + const disk = readDiskVersionCache(); + if (disk) { + cachedVersion = disk.version; + cachedAt = now; + // Stale-while-revalidate (oakimov): always serve disk cache; refresh in + // background when fresh (keep warm) or stale. + scheduleRemoteVersionRefresh(); + return cachedVersion; + } + + scheduleRemoteVersionRefresh(); return CURSOR_AGENT_CLI_VERSION; } +/** + * Await a remote installer scrape (tests / warm-up). Writes disk cache on success. + */ +export async function refreshCursorAgentCliVersionFromInstaller(): Promise { + try { + const id = await fetchInstallerVersionId(); + if (id) { + writeDiskVersionCache({ version: id, fetchedAt: Date.now() }); + cachedVersion = id; + cachedAt = Date.now(); + return id; + } + } catch { + /* ignore */ + } + return null; +} + /** Exposed for testing: reset the in-memory cache. */ export function resetCursorAgentCliVersionCache(): void { cachedVersion = null; cachedAt = 0; + remoteRefreshInFlight = null; + remoteRefreshScheduled = false; +} + +/** Exposed for testing: inject fetch + cache dir. */ +export function configureCursorAgentCliVersionForTests(options: { + fetchImpl?: typeof fetch; + cacheDir?: string | null; +}): void { + if (options.fetchImpl) fetchImpl = options.fetchImpl; + if (options.cacheDir !== undefined) cacheDirOverride = options.cacheDir; +} + +export function resetCursorAgentCliVersionTestHooks(): void { + fetchImpl = fetch; + cacheDirOverride = null; + resetCursorAgentCliVersionCache(); } diff --git a/open-sse/utils/cursorAgentProtobuf.ts b/open-sse/utils/cursorAgentProtobuf.ts index d86fc7f4ce6..21164b6ed10 100644 --- a/open-sse/utils/cursorAgentProtobuf.ts +++ b/open-sse/utils/cursorAgentProtobuf.ts @@ -285,6 +285,12 @@ const CURSOR_MODEL_ALIASES: Record = { "composer-2-5-fast": "composer-2.5-fast", "composer-2.5-sdk-fast": "composer-2.5-fast", "composer-latest-fast": "composer-2.5-fast", + "grok-4.5-medium": "cursor-grok-4.5-medium", + "grok-4.5-fast-medium": "cursor-grok-4.5-medium-fast", + "grok-4.5-high": "cursor-grok-4.5-high", + "grok-4.5-fast-high": "cursor-grok-4.5-high-fast", + "grok-4.5-xhigh": "cursor-grok-4.5-xhigh", + "grok-4.5-fast-xhigh": "cursor-grok-4.5-xhigh-fast", }; export function normalizeCursorModelId(modelId: string): string { @@ -301,6 +307,10 @@ export function normalizeCursorModelId(modelId: string): string { // {id:"reasoning", value:}. "-fast"/"-thinking" are separate toggles // (already handled elsewhere / not covered by this suffix set) and must not // be misread as an effort value. +// +// Grok (`cursor-grok-*` / legacy `grok-*`) follows the Claude-style `effort` +// parameter. Without the split, ids like `cursor-grok-4.5-high` return empty +// turns (same symptom as #7289). Combined `-high-fast` is supported. const CURSOR_EFFORT_SUFFIXES = ["low", "medium", "high", "xhigh", "max"] as const; /** @@ -329,6 +339,40 @@ function splitCursorEffortSuffix( return null; } +/** + * Grok family: strip optional `-fast`, then effort suffix → ModelParameters. + * Prefer `cursor-grok-` over bare `grok-` so `cursor-grok-*` is not mis-matched. + */ +function resolveGrokRequestedModel( + normalized: string +): { modelId: string; parameters: Array<{ id: string; value: string }> } | null { + const prefix = normalized.startsWith("cursor-grok-") + ? "cursor-grok-" + : normalized.startsWith("grok-") + ? "grok-" + : null; + if (!prefix) return null; + + let id = normalized; + const extraParams: Array<{ id: string; value: string }> = []; + if (id.endsWith("-fast") && id.length > prefix.length + "-fast".length) { + id = id.slice(0, -"-fast".length); + extraParams.push({ id: "fast", value: "true" }); + } + + const effortSplit = splitCursorEffortSuffix(id, prefix, "effort"); + if (effortSplit) { + return { + modelId: effortSplit.modelId, + parameters: [...effortSplit.parameters, ...extraParams], + }; + } + if (extraParams.length > 0) { + return { modelId: id, parameters: extraParams }; + } + return null; +} + /** * cursor-agent rewrites model ids before putting them on the wire: * "auto" → RequestedModel { model_id: "default" } @@ -340,6 +384,8 @@ function splitCursorEffortSuffix( * parameters: [{id: "effort", value: "high"}] } * "gpt-5.5-high" → RequestedModel { model_id: "gpt-5.5", * parameters: [{id: "reasoning", value: "high"}] } + * "cursor-grok-4.5-high" → RequestedModel { model_id: "cursor-grok-4.5", + * parameters: [{id: "effort", value: "high"}] } * * Other ids are passed through verbatim after spelling-variant normalization * (see normalizeCursorModelId). @@ -398,6 +444,10 @@ export function resolveRequestedModel( parameters: [{ id: "fast", value: "true" }], }; } + const grokSplit = resolveGrokRequestedModel(normalized); + if (grokSplit) { + return grokSplit; + } const claudeSplit = splitCursorEffortSuffix(normalized, "claude-", "effort"); if (claudeSplit) { return claudeSplit; diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index 11a6f4e779c..9adb5dbf2f6 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -573,7 +573,10 @@ export function createNoopAbortWritable(): { * streaming twin of the non-streaming `isEmptyContentResponse` check. Only * applies to bodies that actually looked like SSE, and terminal states where * emptiness is legitimate (length / tool_calls / content_filter / max_tokens / - * tool_use) are excluded by the watcher. + * tool_use) are excluded by the watcher. If the stream already carried a + * substantive SSE `error` / `response.failed` / Claude `event:error`, stand + * down — same spirit as Claude #3685 `lifecycle.hasError` and readiness #8972 + * (do not invent empty content on top of an actionable error). */ type SilentCloseOutcome = { kind: "truncated" } | { kind: "error"; reason: string }; @@ -586,10 +589,7 @@ function resolveSilentCloseOutcome(input: { if (!input.bytesWereForwarded) return null; if (!input.clientTerminalSeen) { - if ( - input.clientResponseFormat === FORMATS.CLAUDE && - input.contentWatcher.sawContent() - ) { + if (input.clientResponseFormat === FORMATS.CLAUDE && input.contentWatcher.sawContent()) { // #7699 — upstream dropped after content reached the client on a Claude // stream. Keep the partial response: emit a clean max_tokens completion // instead of an error frame so Anthropic SDK / Claude Code don't report @@ -608,9 +608,20 @@ function resolveSilentCloseOutcome(input: { if (input.clientResponseFormat === FORMATS.OPENAI && input.contentWatcher.sawContent()) { return { kind: "error", reason: "Upstream stream ended without a terminal marker" }; } + // Responses-format clients (Codex CLI and other /v1/responses consumers): + // a healthy OpenAI Responses stream ALWAYS terminates with an explicit + // `response.completed` event — it is the format's only terminal marker and + // carries the final status/usage. Content forwarded without it is an + // upstream drop, the same class as #10443 for chat completions; surface a + // synthetic response.failed instead of a silent close so clients report + // the break instead of waiting on a completion event that never comes. + if (isResponsesClientFormat(input.clientResponseFormat) && input.contentWatcher.sawContent()) { + return { kind: "error", reason: "Upstream stream ended without a terminal marker" }; + } } const watcher = input.contentWatcher; + if (watcher.sawError()) return null; if (watcher.sawSseFrame() && !watcher.sawContent() && !watcher.sawLegitEmptyTerminal()) { return { kind: "error", reason: "Provider returned empty content" }; } @@ -672,13 +683,15 @@ export function createDisconnectAwareStream(transformStream, streamController) { if (clientTerminalSeen) return; terminalTail += terminalDecoder.decode(chunk, { stream: true }); - if (terminalTail.length > 4096) { - terminalTail = terminalTail.slice(-4096); - } + // Scan before bounding retained state: a compaction terminal frame can + // exceed the tail budget because encrypted_content is carried inline. clientTerminalSeen = hasClientTerminalSseMarker( terminalTail, streamController.clientResponseFormat ); + if (terminalTail.length > 4096) { + terminalTail = terminalTail.slice(-4096); + } if (clientTerminalSeen) { streamController.markClientTerminalSeen?.(); } diff --git a/open-sse/utils/streamPayloadCollector.ts b/open-sse/utils/streamPayloadCollector.ts index 9ca089dfe4e..31b9e818f46 100644 --- a/open-sse/utils/streamPayloadCollector.ts +++ b/open-sse/utils/streamPayloadCollector.ts @@ -128,6 +128,75 @@ function tryParseJson(raw: string): unknown { } } +/** + * Splits a tool_call `arguments` string that is actually multiple back-to-back JSON + * objects glued together with no separator, into its individual object substrings. + * + * Root cause (observed on opencode/muse-spark-1.2-contributor-free via the zen + * provider): some upstreams never vary `index`/`id` across a 2nd/3rd/… tool_call of + * the SAME name emitted in one turn, so every delta in `buildOpenAISummary` above + * resolves to the same accumulator key and `arguments` ends up as N JSON objects + * concatenated with no delimiter — invalid as a single JSON value, but each object is + * individually well-formed. Structural, not provider-specific: applies to whichever + * upstream exhibits the same index-collision streaming bug. + * + * Returns `null` when `raw` is empty, already valid single JSON, or does not scan as + * ≥2 back-to-back valid JSON values — callers must leave `arguments` untouched in + * that case (never regress a value that used to reach the client as-is). + */ +export function splitConcatenatedToolCallArguments(raw: string): string[] | null { + if (!raw) return null; + try { + JSON.parse(raw); + return null; // Already a single valid JSON value — nothing to split. + } catch { + // Fall through to the multi-value scan below. + } + + const parts: string[] = []; + let depth = 0; + let inString = false; + let escaped = false; + let start = -1; + + for (let i = 0; i < raw.length; i++) { + const ch = raw[i]; + if (start === -1) { + if (ch === " " || ch === "\n" || ch === "\r" || ch === "\t") continue; + if (ch !== "{" && ch !== "[") return null; // Not a value boundary — bail, leave untouched. + start = i; + } + if (inString) { + if (escaped) escaped = false; + else if (ch === "\\") escaped = true; + else if (ch === '"') inString = false; + continue; + } + if (ch === '"') { + inString = true; + continue; + } + if (ch === "{" || ch === "[") depth++; + else if (ch === "}" || ch === "]") { + depth--; + if (depth === 0) { + parts.push(raw.slice(start, i + 1)); + start = -1; + } + } + } + if (start !== -1 || depth !== 0 || parts.length < 2) return null; + + for (const part of parts) { + try { + JSON.parse(part); + } catch { + return null; // One of the scanned segments isn't valid JSON — bail entirely. + } + } + return parts; +} + // ─── Per-format live reducers ──────────────────────────────────────────────── // Each reducer mirrors the corresponding build*Summary()'s original for-loop // body exactly (ingest = one loop iteration, finalize = the post-loop return), @@ -262,7 +331,28 @@ function createOpenAIReducer(fallbackModel?: string | null): SummaryReducer { message.reasoning_content = joinedReasoning; } - const finalToolCalls = [...toolCalls.values()].sort((a, b) => a.index - b.index); + const mergedToolCalls = [...toolCalls.values()].sort((a, b) => a.index - b.index); + // Expand any entry whose accumulated `arguments` turned out to be multiple + // concatenated JSON objects (upstream never varied index/id across repeated + // same-name tool_calls) into its own separate tool_calls entries. + const finalToolCalls: ToolCall[] = []; + let nextIndex = 0; + // Normalize tool_call indexes to contiguous 0-based (OpenAI contract). + for (const tc of mergedToolCalls) { + const splitArgs = splitConcatenatedToolCallArguments(tc.function.arguments); + if (!splitArgs) { + finalToolCalls.push({ ...tc, index: nextIndex++ }); + continue; + } + for (const [i, args] of splitArgs.entries()) { + finalToolCalls.push({ + id: tc.id ? `${tc.id}_split${i}` : null, + index: nextIndex++, + type: tc.type, + function: { name: tc.function.name, arguments: args }, + }); + } + } if (finalToolCalls.length > 0) { finishReason = "tool_calls"; message.tool_calls = finalToolCalls; diff --git a/open-sse/utils/streamReadiness.ts b/open-sse/utils/streamReadiness.ts index 23f57678e72..1696a7a5c52 100644 --- a/open-sse/utils/streamReadiness.ts +++ b/open-sse/utils/streamReadiness.ts @@ -34,6 +34,14 @@ function hasUsefulValue(value: unknown): boolean { if (Array.isArray(value)) return value.some(hasUsefulValue); if (!isRecord(value)) return false; + // A Responses compaction item IS the turn's output: remote compaction + // completes with output = [{type:"compaction", encrypted_content}] and no + // assistant text. Deliberately NOT a blanket encrypted_content key — an + // encrypted reasoning item alone is not user-visible output and must keep + // tripping the #8649 empty-content guard. + // This shape is specific to Responses streams; chat-completion frames do not produce it. + if (value.type === "compaction" && hasNonEmptyString(value.encrypted_content)) return true; + for (const key of [ "content", "text", @@ -167,6 +175,59 @@ const TERMINAL_REASON_PATTERN = /"(?:finish_reason|stop_reason)"\s*:\s*"([^"]+)" const SSE_FIELD_LINE = /(?:^|\r?\n)\s*(?:data|event):/; +/** Same spirit as combo `isSubstantiveError` — non-empty string or non-empty object. */ +function isSubstantiveErrorValue(value: unknown): boolean { + if (value === null || value === undefined) return false; + if (typeof value === "string") return value.trim().length > 0; + if (typeof value === "object" && !Array.isArray(value)) { + const record = value as Record; + if (hasNonEmptyString(record.message)) return true; + return Object.keys(record).length > 0; + } + return value === true; +} + +/** + * True when an SSE frame already carries a structured upstream/client error + * (OpenAI `error`, Claude `event:error` / `type:error`, Responses `response.failed`). + * Used by #8649 so we do not invent "Provider returned empty content" after an + * executor already emitted an actionable error (Claude #3685 / readiness #8972 parity). + */ +export function frameHasStructuredStreamError(frame: string): boolean { + const lines = frame.split(/\r?\n/); + let eventType = ""; + + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed || trimmed.startsWith(":")) continue; + if (trimmed.startsWith("event:")) { + eventType = trimmed.slice(6).trim(); + if (/^error$/i.test(eventType)) return true; + continue; + } + if (!trimmed.startsWith("data:")) continue; + + const data = trimmed.slice(5).trim(); + if (!data || data === "[DONE]") continue; + + try { + const parsed: unknown = JSON.parse(data); + if (!isRecord(parsed)) continue; + const type = getPayloadType(parsed, eventType); + if (type === "error" || type === "response.failed" || eventType === "response.failed") { + return true; + } + if (isSubstantiveErrorValue(parsed.error)) return true; + const nestedResponse = isRecord(parsed.response) ? parsed.response : null; + if (nestedResponse?.status === "failed" && nestedResponse.error != null) return true; + } catch { + // non-JSON data lines are not structured errors + } + } + + return false; +} + export type StreamContentWatcher = { /** Feed a decoded slice of the client-facing stream. Safe to call with partial frames. */ note: (text: string) => void; @@ -183,6 +244,11 @@ export type StreamContentWatcher = { * so callers must not read emptiness into it. */ sawSseFrame: () => boolean; + /** + * True once a substantive SSE error frame was seen. Separate from sawContent + * so #8649 can stand down without treating errors as model output. + */ + sawError: () => boolean; }; /** @@ -195,6 +261,9 @@ export type StreamContentWatcher = { * single frame larger than the cap is scanned in pieces, which can only ever * lose content-detection precision in the direction of "saw content", never * toward a false empty. + * + * Also tracks `sawError` so an already-emitted structured error is not rewritten + * as empty content (parity with Claude #3685 `lifecycle.hasError` and readiness #8972). */ export function createStreamContentWatcher(): StreamContentWatcher { const MAX_BUFFERED = 64 * 1024; @@ -202,10 +271,12 @@ export function createStreamContentWatcher(): StreamContentWatcher { let content = false; let legitEmpty = false; let sse = false; + let error = false; const inspect = (frame: string): void => { if (!frame) return; if (!sse && SSE_FIELD_LINE.test(frame)) sse = true; + if (!error && frameHasStructuredStreamError(frame)) error = true; if (!content && hasUsefulStreamContent(frame)) content = true; if (legitEmpty) return; for (const match of frame.matchAll(TERMINAL_REASON_PATTERN)) { @@ -238,6 +309,7 @@ export function createStreamContentWatcher(): StreamContentWatcher { sawContent: () => content, sawLegitEmptyTerminal: () => legitEmpty, sawSseFrame: () => sse, + sawError: () => error, }; } @@ -262,11 +334,7 @@ function processStreamReadinessEvent(state: StreamReadinessSignalState): boolean try { const payload: unknown = JSON.parse(data); - if ( - !state.upstreamDiagnostic && - isRecord(payload) && - isErrorOnlyStructuredPayload(payload) - ) { + if (!state.upstreamDiagnostic && isRecord(payload) && isErrorOnlyStructuredPayload(payload)) { const error = payload.error; const rawMessage = typeof error === "string" diff --git a/open-sse/utils/syncedEffortVariants.ts b/open-sse/utils/syncedEffortVariants.ts index 2a2c7f0d68c..33c5ec8c56b 100644 --- a/open-sse/utils/syncedEffortVariants.ts +++ b/open-sse/utils/syncedEffortVariants.ts @@ -19,17 +19,17 @@ * only when the base model's own `supportedThinkingEfforts` actually declares that tier — * never a blind string match. * - * Skipped entirely for `codex` and `kimi`-owned models: both already own a conflicting - * native `-{effort}` suffix mechanism (`splitCodexReasoningSuffix` / - * `getKimiCodeStaticThinkingPolicy`), so double-registering here would collide with their - * own alias resolution. Also skipped for any model whose id already ends in a token that - * matches a canonical effort value, to avoid colliding with a model that legitimately ends - * in an effort-like token (e.g. a model literally named "...-high"). + * Skipped entirely for `codex`, `kimi`-owned, and GLM (`glm`, `glm-cn`, `glmt`) models: + * they already own conflicting `-{effort}` aliases (`splitCodexReasoningSuffix`, + * `getKimiCodeStaticThinkingPolicy`, or `GlmExecutor::parseGlmEffortTier`), so generating + * another layer here would create invalid nested ids. Also skipped for any model whose id + * already ends in a token that matches a canonical effort value, to avoid colliding with a + * model that legitimately ends in an effort-like token (e.g. a model named "...-high"). */ import { CANONICAL_EFFORT_VALUES } from "@/shared/reasoning/effortStandardization.ts"; -/** Provider ids that already own a native `-{effort}` suffix mechanism — never double-register. */ -export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex"]); +/** Provider ids with dedicated `-{effort}` aliases — never synthesize another suffix layer. */ +export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex", "glm", "glm-cn", "glmt"]); /** Provider-id prefixes covering that mechanism's multiple connection variants (kimi-coding, kimi-coding-apikey). */ const SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES = ["kimi"]; diff --git a/package-lock.json b/package-lock.json index 26da1997dbb..bf77bab602a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -14,12 +14,10 @@ "packages/browser-pool" ], "dependencies": { - "@atjsh/llmlingua-2": "3.0.0", "@aws-sdk/client-bedrock-runtime": "^3.1112.0", "@dnd-kit/core": "^6.3.1", "@dnd-kit/sortable": "^10.0.0", "@dnd-kit/utilities": "^3.2.2", - "@huggingface/transformers": "^4.2.0", "@lobehub/icons": "^5.16.0", "@modelcontextprotocol/sdk": "^1.29.0", "@monaco-editor/react": "^4.7.0", @@ -62,7 +60,6 @@ "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", - "onnxruntime-node": "~1.27.0", "open": "^11.0.1", "ora": "^9.4.1", "parse5": "^8.0.1", @@ -110,7 +107,7 @@ "@testing-library/jest-dom": "^7.0.1", "@testing-library/react": "^16.3.2", "@types/better-sqlite3": "^9.6.0", - "@types/bun": "*", + "@types/bun": "latest", "@types/node": "^26.2.0", "@types/react": "^19.2.18", "@types/react-dom": "^19.2.4", @@ -157,9 +154,11 @@ }, "optionalDependencies": { "@atjsh/llmlingua-2": "3.0.0", + "@huggingface/transformers": "^4.2.0", "better-sqlite3": "^13.0.2", "js-tiktoken": "^1.0.20", "keytar": "^7.9.0", + "onnxruntime-node": "1.24.3", "sqlite-vec": "^0.1.9", "tls-client-node": "^0.2.0", "wreq-js": "^3.0.0" @@ -353,9 +352,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -370,9 +366,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -387,9 +380,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -404,9 +394,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -4523,6 +4510,7 @@ "resolved": "https://registry.npmjs.org/@huggingface/jinja/-/jinja-0.5.9.tgz", "integrity": "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw==", "license": "MIT", + "optional": true, "engines": { "node": ">=18" } @@ -4531,13 +4519,15 @@ "version": "0.1.3", "resolved": "https://registry.npmjs.org/@huggingface/tokenizers/-/tokenizers-0.1.3.tgz", "integrity": "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA==", - "license": "Apache-2.0" + "license": "Apache-2.0", + "optional": true }, "node_modules/@huggingface/transformers": { "version": "4.2.0", "resolved": "https://registry.npmjs.org/@huggingface/transformers/-/transformers-4.2.0.tgz", "integrity": "sha512-8BRCoBMH0XsWaEIamuR0LrJGAfftgHAfb2Vrffy0VKlSAE/MnUJ5/h/zTfEP3fDIft+nk7TqB8xXEyABGitBjQ==", "license": "Apache-2.0", + "optional": true, "dependencies": { "@huggingface/jinja": "^0.5.6", "@huggingface/tokenizers": "^0.1.3", @@ -4546,97 +4536,6 @@ "sharp": "^0.34.5" } }, - "node_modules/@huggingface/transformers/node_modules/global-agent": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-3.0.0.tgz", - "integrity": "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q==", - "license": "BSD-3-Clause", - "dependencies": { - "boolean": "^3.0.1", - "es6-error": "^4.1.1", - "matcher": "^3.0.0", - "roarr": "^2.15.3", - "semver": "^7.3.2", - "serialize-error": "^7.0.1" - }, - "engines": { - "node": ">=10.0" - } - }, - "node_modules/@huggingface/transformers/node_modules/matcher": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/matcher/-/matcher-3.0.0.tgz", - "integrity": "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng==", - "license": "MIT", - "dependencies": { - "escape-string-regexp": "^4.0.0" - }, - "engines": { - "node": ">=10" - } - }, - "node_modules/@huggingface/transformers/node_modules/onnxruntime-common": { - "version": "1.24.3", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.3.tgz", - "integrity": "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==", - "license": "MIT" - }, - "node_modules/@huggingface/transformers/node_modules/onnxruntime-node": { - "version": "1.24.3", - "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.24.3.tgz", - "integrity": "sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==", - "hasInstallScript": true, - "license": "MIT", - "os": [ - "win32", - "darwin", - "linux" - ], - "dependencies": { - "adm-zip": "^0.5.16", - "global-agent": "^3.0.0", - "onnxruntime-common": "1.24.3" - } - }, - "node_modules/@huggingface/transformers/node_modules/semver": { - "version": "7.8.5", - "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", - "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", - "license": "ISC", - "bin": { - "semver": "bin/semver.js" - }, - "engines": { - "node": ">=10" - } - }, - "node_modules/@huggingface/transformers/node_modules/serialize-error": { - "version": "7.0.1", - "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-7.0.1.tgz", - "integrity": "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw==", - "license": "MIT", - "dependencies": { - "type-fest": "^0.13.1" - }, - "engines": { - "node": ">=10" - }, - "funding": { - "url": "https://github.com/sponsors/sindresorhus" - } - }, - "node_modules/@huggingface/transformers/node_modules/type-fest": { - "version": "0.13.1", - "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.13.1.tgz", - "integrity": "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg==", - "license": "(MIT OR CC0-1.0)", - "engines": { - "node": ">=10" - }, - "funding": { - "url": "https://github.com/sponsors/sindresorhus" - } - }, "node_modules/@humanfs/core": { "version": "0.19.1", "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", @@ -6562,9 +6461,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -6581,9 +6477,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -6600,9 +6493,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -6619,9 +6509,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8503,9 +8390,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8523,9 +8407,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8543,9 +8424,6 @@ "ppc64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8563,9 +8441,6 @@ "riscv64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8583,9 +8458,6 @@ "riscv64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8603,9 +8475,6 @@ "s390x" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8623,9 +8492,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8643,9 +8509,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8839,9 +8702,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8856,9 +8716,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8873,9 +8730,6 @@ "ppc64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8890,9 +8744,6 @@ "riscv64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8907,9 +8758,6 @@ "riscv64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8924,9 +8772,6 @@ "s390x" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8941,9 +8786,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8958,9 +8800,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -9647,30 +9486,35 @@ "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/aspromise/-/aspromise-1.1.2.tgz", "integrity": "sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/base64": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/base64/-/base64-1.1.2.tgz", "integrity": "sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/codegen": { "version": "2.0.5", "resolved": "https://registry.npmjs.org/@protobufjs/codegen/-/codegen-2.0.5.tgz", "integrity": "sha512-zgXFLzW3Ap33e6d0Wlj4MGIm6Ce8O89n/apUaGNB/jx+hw+ruWEp7EwGUshdLKVRCxZW12fp9r40E1mQrf/34g==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/eventemitter": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz", "integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/fetch": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz", "integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==", + "devOptional": true, "license": "BSD-3-Clause", "dependencies": { "@protobufjs/aspromise": "^1.1.1" @@ -9680,24 +9524,28 @@ "version": "1.0.2", "resolved": "https://registry.npmjs.org/@protobufjs/float/-/float-1.0.2.tgz", "integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/path": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz", "integrity": "sha512-6JOcJ5Tm08dOHAbdR3GrvP+yUUfkjG5ePsHYczMFLq3ZmMkAD98cDgcT2iA1lJ9NVwFd4tH/iSSoe44YWkltEA==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/pool": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/@protobufjs/pool/-/pool-1.1.0.tgz", "integrity": "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@protobufjs/utf8": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/@protobufjs/utf8/-/utf8-1.1.1.tgz", "integrity": "sha512-oOAWABowe8EAbMyWKM0tYDKi8Yaox52D+HWZhAIJqQXbqe0xI/GV7FhLWqlEKreMkfDjshR5FKgi3mnle0h6Eg==", + "devOptional": true, "license": "BSD-3-Clause" }, "node_modules/@radix-ui/number": { @@ -11819,9 +11667,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11838,9 +11683,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11857,9 +11699,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11876,9 +11715,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11895,9 +11731,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11914,9 +11747,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -12919,6 +12749,7 @@ "version": "26.2.0", "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", + "devOptional": true, "license": "MIT", "dependencies": { "undici-types": "~8.3.0" @@ -13999,9 +13830,6 @@ "arm" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -14016,9 +13844,6 @@ "arm" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -14033,9 +13858,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -14050,9 +13872,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -14067,9 +13886,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -14084,9 +13900,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -14198,6 +14011,7 @@ "resolved": "https://registry.npmjs.org/adm-zip/-/adm-zip-0.6.0.tgz", "integrity": "sha512-XleryMhbuksdKtofnWZ9Sk+4CUTbms4Mb/EU32SZwToAyZ5RgVos/ki8n+yr0LWHOGKuakbXTuuYNHLQjhddgg==", "license": "MIT", + "optional": true, "engines": { "node": ">=14.0" } @@ -15171,7 +14985,8 @@ "resolved": "https://registry.npmjs.org/boolean/-/boolean-3.2.0.tgz", "integrity": "sha512-d0II/GO9uf9lfUHH2BQsjxzRJZBdsjgsBiW4BvhWk/3qoKwQFjIDVN19PfX8F2D/r9PCMTtLWjYVCFrpeYUzsw==", "deprecated": "Package no longer supported. Contact Support at https://www.npmjs.com/support for more info.", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/bottleneck": { "version": "2.19.5", @@ -18135,6 +17950,7 @@ "version": "1.1.4", "resolved": "https://registry.npmjs.org/define-data-property/-/define-data-property-1.1.4.tgz", "integrity": "sha512-rBMvIzlpA8v6E+SJZoo++HAYqsLrkg7MSfIinMPFhmkorw7X+dOXVJQs+QT69zGkzMyfDnIMN2Wid1+NbL3T+A==", + "devOptional": true, "license": "MIT", "dependencies": { "es-define-property": "^1.0.0", @@ -18164,6 +17980,7 @@ "version": "1.2.1", "resolved": "https://registry.npmjs.org/define-properties/-/define-properties-1.2.1.tgz", "integrity": "sha512-8QmQKqEASLd5nx0U1B1okLElbUuuttJ/AnYmRXbbbGDWh6uS208EjD4Xqq/I9wK7u0v6O08XhTWnt5XtEbR6Dg==", + "devOptional": true, "license": "MIT", "dependencies": { "define-data-property": "^1.0.1", @@ -18264,7 +18081,8 @@ "version": "2.1.0", "resolved": "https://registry.npmjs.org/detect-node/-/detect-node-2.1.0.tgz", "integrity": "sha512-T0NIuQpnTvFDATNuHN5roPwSBG83rFsuO+MXXH9/3N1eFbn4wcPjttvjMLEPWJ0RGUYgQE7cGgS3tNxbqCGM7g==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/detect-node-es": { "version": "1.1.0", @@ -19108,7 +18926,8 @@ "version": "4.1.1", "resolved": "https://registry.npmjs.org/es6-error/-/es6-error-4.1.1.tgz", "integrity": "sha512-Um/+FxMr9CISWh0bi5Zv0iOD+4cFh5qLeks1qhAopKVAJw3drgKbKySikp7wGhDL0HPeaja0P5ULZrxLkniUVg==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/es6-promisify": { "version": "7.0.0", @@ -20600,7 +20419,8 @@ "version": "25.9.23", "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-25.9.23.tgz", "integrity": "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ==", - "license": "Apache-2.0" + "license": "Apache-2.0", + "optional": true }, "node_modules/flatted": { "version": "3.4.2", @@ -21422,15 +21242,18 @@ } }, "node_modules/global-agent": { - "version": "4.1.3", - "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-4.1.3.tgz", - "integrity": "sha512-KUJEViiuFT3I97t+GYMikLPJS2Lfo/S2F+DQuBWzuzaMPnvt5yyZePzArx36fBzpGTxZjIpDbXLeySLgh+k76g==", + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-3.0.0.tgz", + "integrity": "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q==", "license": "BSD-3-Clause", + "optional": true, "dependencies": { - "globalthis": "^1.0.2", - "matcher": "^4.0.0", - "semver": "^7.3.5", - "serialize-error": "^8.1.0" + "boolean": "^3.0.1", + "es6-error": "^4.1.1", + "matcher": "^3.0.0", + "roarr": "^2.15.3", + "semver": "^7.3.2", + "serialize-error": "^7.0.1" }, "engines": { "node": ">=10.0" @@ -21441,6 +21264,7 @@ "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", "license": "ISC", + "optional": true, "bin": { "semver": "bin/semver.js" }, @@ -21489,6 +21313,7 @@ "version": "1.0.4", "resolved": "https://registry.npmjs.org/globalthis/-/globalthis-1.0.4.tgz", "integrity": "sha512-DpLKbNU4WylpxJykQujfCcwYWiV/Jhm50Goo0wrVILAv5jOr9d+H+UR3PhSCD2rCCEIg0uc+G+muBTwD54JhDQ==", + "devOptional": true, "license": "MIT", "dependencies": { "define-properties": "^1.2.1", @@ -21843,7 +21668,8 @@ "version": "1.0.9", "resolved": "https://registry.npmjs.org/guid-typescript/-/guid-typescript-1.0.9.tgz", "integrity": "sha512-Y8T4vYhEfwJOTbouREvG+3XDsjr8E3kIr7uf+JZ0BYloFsttiHU0WfvANVsR7TxNUJa/WpCnw/Ino/p+DeBhBQ==", - "license": "ISC" + "license": "ISC", + "optional": true }, "node_modules/hachure-fill": { "version": "0.5.2", @@ -21877,6 +21703,7 @@ "version": "1.0.2", "resolved": "https://registry.npmjs.org/has-property-descriptors/-/has-property-descriptors-1.0.2.tgz", "integrity": "sha512-55JNKuIW+vq4Ke1BjOTjM2YctQIvCT7GFzHwmfZPGo5wnrgkid0YQtnAleFSqumZm4az3n2BS+erby5ipJdgrg==", + "devOptional": true, "license": "MIT", "dependencies": { "es-define-property": "^1.0.0" @@ -24988,7 +24815,8 @@ "version": "5.0.1", "resolved": "https://registry.npmjs.org/json-stringify-safe/-/json-stringify-safe-5.0.1.tgz", "integrity": "sha512-ZClg6AaYvamvYEE82d3Iyd3vSSIjQ+odgjaTzRuO3s7toCdFKczob2i0zCh7JE8kWn17yvAWhUVxvqGwUalsRA==", - "license": "ISC" + "license": "ISC", + "optional": true }, "node_modules/json5": { "version": "2.2.3", @@ -26848,18 +26676,16 @@ } }, "node_modules/matcher": { - "version": "4.0.0", - "resolved": "https://registry.npmjs.org/matcher/-/matcher-4.0.0.tgz", - "integrity": "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ==", + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/matcher/-/matcher-3.0.0.tgz", + "integrity": "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng==", "license": "MIT", + "optional": true, "dependencies": { "escape-string-regexp": "^4.0.0" }, "engines": { "node": ">=10" - }, - "funding": { - "url": "https://github.com/sponsors/sindresorhus" } }, "node_modules/material-symbols": { @@ -29626,6 +29452,7 @@ "version": "1.1.1", "resolved": "https://registry.npmjs.org/object-keys/-/object-keys-1.1.1.tgz", "integrity": "sha512-NuAESUOUMrlIXOfHKzD6bpPu3tYt3xvjNdRIQ+FeT0lNb4K8WR70CaDxhuNguS2XG+GjkyMwOzsN5ZktImfhLA==", + "devOptional": true, "license": "MIT", "engines": { "node": ">= 0.4" @@ -29830,17 +29657,19 @@ } }, "node_modules/onnxruntime-common": { - "version": "1.27.0", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.27.0.tgz", - "integrity": "sha512-3KxL5wIVqa8Ex08jxSzncm9CMgw8CjOFyOQ7SxvG9o0cVLlhTNKXyIQuTbtX4tGPJEf73OER2xrjt4HJSBL4ow==", - "license": "MIT" + "version": "1.24.3", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.3.tgz", + "integrity": "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==", + "license": "MIT", + "optional": true }, "node_modules/onnxruntime-node": { - "version": "1.27.0", - "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.27.0.tgz", - "integrity": "sha512-QEzGwrvNBgv4uPVdnbHsOGG4G6T96mdlcFI8aAKPjMU8wOPpVocPXb6k3QGkaZagVTv2G9Bnnbo6Z3JdXr1fQw==", + "version": "1.24.3", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.24.3.tgz", + "integrity": "sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==", "hasInstallScript": true, "license": "MIT", + "optional": true, "os": [ "win32", "darwin", @@ -29848,8 +29677,8 @@ ], "dependencies": { "adm-zip": "^0.5.16", - "global-agent": "^4.1.3", - "onnxruntime-common": "1.27.0" + "global-agent": "^3.0.0", + "onnxruntime-common": "1.24.3" } }, "node_modules/onnxruntime-web": { @@ -29857,6 +29686,7 @@ "resolved": "https://registry.npmjs.org/onnxruntime-web/-/onnxruntime-web-1.26.0-dev.20260416-b7804b056c.tgz", "integrity": "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw==", "license": "MIT", + "optional": true, "dependencies": { "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", @@ -29870,13 +29700,15 @@ "version": "5.3.2", "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", - "license": "Apache-2.0" + "license": "Apache-2.0", + "optional": true }, "node_modules/onnxruntime-web/node_modules/onnxruntime-common": { "version": "1.24.0-dev.20251116-b39e144322", "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.0-dev.20251116-b39e144322.tgz", "integrity": "sha512-BOoomdHYmNRL5r4iQ4bMvsl2t0/hzVQ3OM3PHD0gxeXu1PmggqBv3puZicEUVOA3AtHHYmqZtjMj9FOfGrATTw==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/open": { "version": "11.0.1", @@ -30023,9 +29855,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -30065,9 +29894,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -30081,9 +29907,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -31116,7 +30939,8 @@ "version": "1.3.6", "resolved": "https://registry.npmjs.org/platform/-/platform-1.3.6.tgz", "integrity": "sha512-fnWVljUchTro6RiCFvCXBbNhJc2NijN7oIQxbwsyL0buWJPG85v81ehlHI9fXrJsMNgTofEoWIQeClKpgxFLrg==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/playwright": { "version": "1.62.1", @@ -32071,6 +31895,7 @@ "version": "7.6.5", "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.5.tgz", "integrity": "sha512-/FPD0nUc9jH6rfFjji9IBqOz4pcSE3CsT1m7Ep6Mdb0LxSUMj8hgl6GomOvZzpNpAqqGaXA0P3VSrZLFzIhQrw==", + "devOptional": true, "hasInstallScript": true, "license": "BSD-3-Clause", "dependencies": { @@ -32094,6 +31919,7 @@ "version": "5.3.2", "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", + "devOptional": true, "license": "Apache-2.0" }, "node_modules/proxy-addr": { @@ -33490,6 +33316,7 @@ "resolved": "https://registry.npmjs.org/roarr/-/roarr-2.15.4.tgz", "integrity": "sha512-CHhPh+UNHD2GTXNYhPWLnU8ONHdI+5DI+4EYIAOaiD63rHeYlZvyh8P+in5999TTSFgUYuKUAjzRI4mdh/p+2A==", "license": "BSD-3-Clause", + "optional": true, "dependencies": { "boolean": "^3.0.1", "detect-node": "^2.0.4", @@ -33865,7 +33692,8 @@ "version": "1.0.0", "resolved": "https://registry.npmjs.org/semver-compare/-/semver-compare-1.0.0.tgz", "integrity": "sha512-YM3/ITh2MJ5MtzaM429anh+x2jiLVjqILF4m4oyQB18W7Ggea7BfqdH/wGMK7dDiMghv/6WG7znWMwUDzJiXow==", - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/send": { "version": "1.2.1", @@ -33894,12 +33722,13 @@ } }, "node_modules/serialize-error": { - "version": "8.1.0", - "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-8.1.0.tgz", - "integrity": "sha512-3NnuWfM6vBYoy5gZFvHiYsVbafvI9vZv/+jlIigFn4oP4zjNPK3LhcY0xSCgeb1a5L8jO71Mit9LlNoi2UfDDQ==", + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-7.0.1.tgz", + "integrity": "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw==", "license": "MIT", + "optional": true, "dependencies": { - "type-fest": "^0.20.2" + "type-fest": "^0.13.1" }, "engines": { "node": ">=10" @@ -33909,10 +33738,11 @@ } }, "node_modules/serialize-error/node_modules/type-fest": { - "version": "0.20.2", - "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.20.2.tgz", - "integrity": "sha512-Ne+eE4r0/iWnpAxD852z3A+N0Bt5RN//NjJwRd2VFHEmrywxf5vsZlh4R6lixl6B+wz/8d+maTSAkN1FIkI3LQ==", + "version": "0.13.1", + "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.13.1.tgz", + "integrity": "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg==", "license": "(MIT OR CC0-1.0)", + "optional": true, "engines": { "node": ">=10" }, @@ -34684,7 +34514,8 @@ "version": "1.1.3", "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.1.3.tgz", "integrity": "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA==", - "license": "BSD-3-Clause" + "license": "BSD-3-Clause", + "optional": true }, "node_modules/sql.js": { "version": "1.14.2", @@ -36397,6 +36228,7 @@ "version": "8.3.0", "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz", "integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==", + "devOptional": true, "license": "MIT" }, "node_modules/unicode-emoji-modifier-base": { diff --git a/package.json b/package.json index 6caf3dc373c..55cc574f9ec 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "omniroute", "version": "3.8.50", - "description": "Unified AI router with 346 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", + "description": "Unified AI router with 348 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { "omniroute": "bin/omniroute.mjs", @@ -337,18 +337,18 @@ "xxhash-wasm": "^1.1.0", "yazl": "^3.3.1", "zod": "^4.4.3", - "zustand": "^5.0.15", - "@huggingface/transformers": "^4.2.0", - "onnxruntime-node": "~1.27.0" + "zustand": "^5.0.15" }, "optionalDependencies": { "@atjsh/llmlingua-2": "3.0.0", + "@huggingface/transformers": "^4.2.0", "better-sqlite3": "^13.0.2", "js-tiktoken": "^1.0.20", "keytar": "^7.9.0", + "onnxruntime-node": "1.24.3", + "sqlite-vec": "^0.1.9", "tls-client-node": "^0.2.0", - "wreq-js": "^3.0.0", - "sqlite-vec": "^0.1.9" + "wreq-js": "^3.0.0" }, "devDependencies": { "@axe-core/playwright": "^4.13.0", @@ -432,6 +432,7 @@ "unrs-resolver": true }, "overrides": { + "onnxruntime-node": "1.24.3", "fast-xml-parser": "^5.10.1", "sharp": "^0.35.3", "postcss": "^8.5.18", diff --git a/scripts/ad-hoc/dry-run-strict-zero-cost.ts b/scripts/ad-hoc/dry-run-strict-zero-cost.ts new file mode 100644 index 00000000000..54ea41f4d39 --- /dev/null +++ b/scripts/ad-hoc/dry-run-strict-zero-cost.ts @@ -0,0 +1,105 @@ +/** + * Ad-hoc, one-shot dry run of STRICT_ZERO_COST against the real candidate + * pools currently served by this OmniRoute instance (fetched via the + * existing read-only `GET /v1/auto-combo/{channel}/candidates` endpoint — + * no changes made, no billable calls). Not wired into any test suite. + * + * Simulates the filter offline: no live usage-quota state is available + * (that adapter only runs inside the deployed container), so + * `resolveFreeAccessState` always returns `undefined` here — meaning any + * quota-based candidate is reported UNKNOWN unless it lacks even a usage + * adapter, in which case it's reported UNKNOWN for that reason instead. This + * intentionally shows the current, honest ceiling of what's usable today. + * + * Uses each candidate's REAL `connectionId` from the live endpoint (rather + * than assuming) to also exercise the post-code-review connection-safety + * check: a `keyless`-catalogued model whose live `connectionId` is NOT the + * no-auth sentinel is correctly reported as excluded here too. + */ +import { readFileSync } from "node:fs"; +import { + evaluateCandidateConnections, + findBudgetEntry, +} from "../../open-sse/services/autoCombo/strictZeroCostFilter.ts"; +import { SYNTHETIC_NOAUTH_CONNECTION_ID } from "../../open-sse/services/autoCombo/resilienceCandidateFilter.ts"; +import { USAGE_FETCHER_PROVIDERS } from "../../open-sse/services/usage.ts"; + +const usageProviders = new Set(USAGE_FETCHER_PROVIDERS); +const OPTIONS = { minRemainingAllowance: 1, maxStateAgeMs: 180_000 }; + +interface Candidate { + provider: string; + model: string; + connectionId: string; +} + +function loadCandidates(path: string): Candidate[] { + const raw = JSON.parse(readFileSync(path, "utf8")); + const list = Array.isArray(raw) ? raw : raw.candidates; + // The candidates endpoint's `model` field is the FULL "/" + // string (`modelStr` — the leading segment is sometimes the provider id, + // e.g. "groq/...", sometimes its short alias, e.g. "oc/..." for opencode); + // FREE_MODEL_BUDGETS.modelId is always bare. Strip exactly the first "/" + // segment (whichever form it is) so e.g. "groq/meta-llama/llama-4-scout..." + // becomes "meta-llama/llama-4-scout..." and "oc/big-pickle" becomes + // "big-pickle", matching the catalog's modelId either way. + return list.map((c: { provider: string; model: string; connectionId?: string }) => { + const slash = c.model.indexOf("/"); + return { + provider: c.provider, + model: slash === -1 ? c.model : c.model.slice(slash + 1), + connectionId: c.connectionId ?? SYNTHETIC_NOAUTH_CONNECTION_ID, + }; + }); +} + +function run(label: string, path: string): void { + const candidates = loadCandidates(path); + console.log(`\n=== ${label} — ${candidates.length} candidati live ===`); + + const kept: Candidate[] = []; + const excluded: { candidate: Candidate; reason: string }[] = []; + + for (const c of candidates) { + const entry = findBudgetEntry(c); + if (!entry) { + excluded.push({ candidate: c, reason: "non presente nel catalogo free curato" }); + continue; + } + const isNoAuthConnection = c.connectionId === SYNTHETIC_NOAUTH_CONNECTION_ID; + if (entry.freeType === "keyless") { + const safe = evaluateCandidateConnections(c, entry, () => undefined, OPTIONS); + if (safe.length > 0) { + kept.push(c); + } else if (!isNoAuthConnection) { + excluded.push({ + candidate: c, + reason: + "keyless nel catalogo ma raggiunto tramite una connessione DB reale (non il sentinel noauth) — shortcut non applicato, richiederebbe hardStopGuaranteed", + }); + } else { + excluded.push({ candidate: c, reason: "keyless ma valutazione fallita (inatteso)" }); + } + continue; + } + const hasAdapter = usageProviders.has(entry.provider); + const reason = !hasAdapter + ? `nessun usage adapter per '${entry.provider}' in USAGE_FETCHER_PROVIDERS` + : entry.hardStopGuaranteed !== true + ? "hardStopGuaranteed non dichiarato per questo modello" + : "nessuno stato quota live disponibile in questo dry-run offline (richiederebbe il container reale)"; + excluded.push({ candidate: c, reason }); + } + + console.log(`PRIMA (STRICT_ZERO_COST off): ${candidates.length} candidati`); + console.log(`DOPO (STRICT_ZERO_COST on): ${kept.length} candidati sopravvissuti`); + console.log("Sopravvissuti:"); + for (const c of kept) console.log(` OK ${c.provider}/${c.model}`); + console.log("Esclusi (motivo):"); + for (const { candidate: c, reason } of excluded) { + console.log(` EXCL ${c.provider}/${c.model} — ${reason}`); + } +} + +run("auto/coding:free", process.argv[2] ?? "/tmp/dryrun_coding_free.json"); +run("auto/best-free", process.argv[3] ?? "/tmp/dryrun_best-free.json"); diff --git a/scripts/ad-hoc/sync-cursor-models.mjs b/scripts/ad-hoc/sync-cursor-models.mjs index 48698d016ad..62f30bb718e 100644 --- a/scripts/ad-hoc/sync-cursor-models.mjs +++ b/scripts/ad-hoc/sync-cursor-models.mjs @@ -1,12 +1,11 @@ #!/usr/bin/env node -// Sync the cursor models list in open-sse/config/providerRegistry.ts from -// cursor-agent's runtime model list. Triggers an intentional invalid --model -// invocation so cursor-agent prints "Available models: ..." on stderr. +// Sync the cursor models list in open-sse/config/providers/registry/cursor/index.ts +// from cursor-agent's runtime model list (`--list-models`). // // Usage: // node scripts/ad-hoc/sync-cursor-models.mjs # spawn cursor-agent and apply // node scripts/ad-hoc/sync-cursor-models.mjs --dry-run # print proposed block, don't write -// node scripts/ad-hoc/sync-cursor-models.mjs --from-stdin # read the error message from stdin +// node scripts/ad-hoc/sync-cursor-models.mjs --from-stdin # read --list-models output from stdin import { spawnSync } from "node:child_process"; import { readFileSync, writeFileSync } from "node:fs"; @@ -14,7 +13,17 @@ import { fileURLToPath } from "node:url"; import { dirname, resolve } from "node:path"; const __dirname = dirname(fileURLToPath(import.meta.url)); -const REGISTRY_PATH = resolve(__dirname, "..", "open-sse", "config", "providerRegistry.ts"); +const REGISTRY_PATH = resolve( + __dirname, + "..", + "..", + "open-sse", + "config", + "providers", + "registry", + "cursor", + "index.ts" +); const args = new Set(process.argv.slice(2)); const DRY_RUN = args.has("--dry-run"); diff --git a/scripts/build/assembleStandalone.mjs b/scripts/build/assembleStandalone.mjs index b4c8d12c1fa..ee8d730ccf2 100644 --- a/scripts/build/assembleStandalone.mjs +++ b/scripts/build/assembleStandalone.mjs @@ -628,12 +628,11 @@ function copyNativeAssetsAndExtraModules(projectRoot, resolvedOutDir) { * This keeps the fix narrowly scoped to packages the standalone already expects. * * @param {string} projectRoot - * @param {string} resolvedOutDir + * @param {string} bundleNodeModules * @returns {{repaired: number, packages: string[]}} */ -function repairEmptyExternalPackageDirs(projectRoot, resolvedOutDir) { +function repairEmptyExternalPackageDirs(projectRoot, bundleNodeModules) { const summary = { repaired: 0, packages: [] }; - const bundleNodeModules = path.join(resolvedOutDir, "node_modules"); const sourceNodeModules = path.join(projectRoot, "node_modules"); if (!fsSync.existsSync(bundleNodeModules) || !fsSync.existsSync(sourceNodeModules)) { return summary; @@ -899,12 +898,23 @@ export function assembleStandalone({ // 6. Optionally copy native assets + extra modules (synchronous) if (copyNatives) { copyNativeAssetsAndExtraModules(projectRoot, resolvedOutDir); - const emptyPkgRepair = repairEmptyExternalPackageDirs(projectRoot, resolvedOutDir); - if (emptyPkgRepair.repaired > 0) { - console.log( - `[assembleStandalone] Repaired ${emptyPkgRepair.repaired} hollow external package dir(s): ` + - emptyPkgRepair.packages.join(", ") - ); + // Repair hollow externalized package dirs in BOTH locations Turbopack's standalone + // tracer can populate: the top-level bundle node_modules, and — for projects with a + // custom distDir (see next.config.mjs) — the nested /node_modules mirrored + // alongside the traced server chunks. materializeBundledSymlinks (step 7 below) already + // treats these as two distinct targets; #9913 only covered the top-level one, which left + // the nested location's hollow dirs unrepaired (#7346). + for (const bundleNodeModules of [ + path.join(resolvedOutDir, "node_modules"), + path.join(resolvedOutDir, relDistDir, "node_modules"), + ]) { + const emptyPkgRepair = repairEmptyExternalPackageDirs(projectRoot, bundleNodeModules); + if (emptyPkgRepair.repaired > 0) { + console.log( + `[assembleStandalone] Repaired ${emptyPkgRepair.repaired} hollow external package dir(s) in ` + + `${path.relative(resolvedOutDir, bundleNodeModules) || "."}: ${emptyPkgRepair.packages.join(", ")}` + ); + } } // #9166: dynamically imported LLMLingua packages are not reliably traced diff --git a/scripts/build/build-next-isolated.mjs b/scripts/build/build-next-isolated.mjs index 37cd31a14ac..2a444174f18 100644 --- a/scripts/build/build-next-isolated.mjs +++ b/scripts/build/build-next-isolated.mjs @@ -131,12 +131,12 @@ function runNextBuild() { } export function resolveNextBuildBundlerFlag(baseEnv = process.env) { - // Turbopack is the default production bundler (Next 16 stable). Benchmarked on - // this codebase: 2-3x faster than the single-threaded webpack pass (17min -> 9min - // on a 32-core box; ~20min -> 7min on ubuntu-latest), artifact validated - // end-to-end (standalone smoke + e2e/package/electron CI jobs). Webpack stays as - // the explicit escape hatch (=0) for bundler-compat regressions. - return baseEnv.OMNIROUTE_USE_TURBOPACK === "0" ? "--webpack" : "--turbopack"; + // Turbopack is the default on Node.js; on Bun or when explicitly disabled (=0), + // use Webpack (--webpack) to avoid Turbopack V8 internal worker API mismatches. + if (process.versions.bun || baseEnv.OMNIROUTE_USE_TURBOPACK === "0") { + return "--webpack"; + } + return "--turbopack"; } /** diff --git a/scripts/check/check-env-doc-sync.mjs b/scripts/check/check-env-doc-sync.mjs index f2959131a01..9520da76472 100644 --- a/scripts/check/check-env-doc-sync.mjs +++ b/scripts/check/check-env-doc-sync.mjs @@ -156,8 +156,11 @@ const IGNORE_FROM_CODE = new Set([ // X11/Wayland display server vars used by tray heuristic (isTraySupported). "DISPLAY", "WAYLAND_DISPLAY", - // Build-time override for OpenAPI spec path used by generate-api-commands.mjs. + // Build-time overrides for generate-api-commands.mjs (spec input / commands output dir). + // OPENAPI_OUT_DIR exists so tests/unit/cli-api-generator-ref-params.test.ts can regenerate + // into a scratch dir instead of the real bin/cli/api-commands/ tree. "OPENAPI_SPEC", + "OPENAPI_OUT_DIR", // Aliases for documented vars handled via fallback ordering. "API_KEY", "APP_URL", diff --git a/scripts/check/check-known-symbols.ts b/scripts/check/check-known-symbols.ts index 301bf4e276c..6537db388cb 100644 --- a/scripts/check/check-known-symbols.ts +++ b/scripts/check/check-known-symbols.ts @@ -263,7 +263,7 @@ export function findNewMcpTools(frozen: readonly string[], live: Set): s * the reason in the commit message. * * Sources: - * - MCP_TOOLS (33 base tools: omniroute_* + compression + agent_skills) + * - MCP_TOOLS (34 base tools: omniroute_* + compression + agent_skills) * - memoryTools (3): omniroute_memory_* * - skillTools (4): omniroute_skills_* * - gamificationTools (8): gamification_* @@ -273,7 +273,7 @@ export function findNewMcpTools(frozen: readonly string[], live: Set): s * agentSkillTools and compressionTools are included in MCP_TOOLS (deduped by RESERVED_MCP_NAMES). */ export const KNOWN_MCP_TOOL_NAMES: readonly string[] = [ - // MCP_TOOLS base (33) + // MCP_TOOLS base (34) "omniroute_get_health", "omniroute_list_combos", "omniroute_get_combo_metrics", @@ -283,6 +283,7 @@ export const KNOWN_MCP_TOOL_NAMES: readonly string[] = [ "omniroute_cost_report", "omniroute_list_models_catalog", "omniroute_web_search", + "omniroute_x_search", "omniroute_simulate_route", "omniroute_set_budget_guard", "omniroute_set_routing_strategy", diff --git a/scripts/check/check-public-creds.mjs b/scripts/check/check-public-creds.mjs index 21fbe3b563e..7e065705d0a 100644 --- a/scripts/check/check-public-creds.mjs +++ b/scripts/check/check-public-creds.mjs @@ -98,7 +98,6 @@ export const KNOWN_LITERAL_CREDS = new Set([ "open-sse/services/usage/minimax.ts:213:minimax", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (getMiniMaxUsage signature) "open-sse/services/usage/minimax.ts:213:minimax-cn", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (getMiniMaxUsage signature) "open-sse/executors/zcodeProtocol.ts:302:omniroute-${process.pid}", // local per-process ZCode handshake ID, not an upstream credential - "open-sse/executors/copilot-m365-web.ts:330:access_token=${result.accessToken}; chathubPath=${chathubPath}", // dynamic header format string in token refresh ]); /** diff --git a/scripts/check/check-supported-node-runtime.ts b/scripts/check/check-supported-node-runtime.ts index 197966822e9..f2b40e53269 100644 --- a/scripts/check/check-supported-node-runtime.ts +++ b/scripts/check/check-supported-node-runtime.ts @@ -15,6 +15,12 @@ if (!support.nodeCompatible) { process.exit(1); } -console.log( - `Node.js ${support.nodeVersion} satisfies OmniRoute secure runtime policy (${support.supportedRange}).` -); +if (process.versions.bun) { + console.log( + `Bun ${process.versions.bun} (${support.nodeVersion}) satisfies OmniRoute secure runtime policy.` + ); +} else { + console.log( + `Node.js ${support.nodeVersion} satisfies OmniRoute secure runtime policy (${support.supportedRange}).` + ); +} diff --git a/scripts/cli/generate-api-commands.mjs b/scripts/cli/generate-api-commands.mjs index b937ceb3895..b0bbff85f7e 100644 --- a/scripts/cli/generate-api-commands.mjs +++ b/scripts/cli/generate-api-commands.mjs @@ -11,7 +11,7 @@ import * as yaml from "js-yaml"; const __dirname = dirname(fileURLToPath(import.meta.url)); const ROOT = join(__dirname, "..", ".."); const SPEC_PATH = process.env.OPENAPI_SPEC || join(ROOT, "docs/openapi.yaml"); -const OUT_DIR = join(ROOT, "bin/cli/api-commands"); +const OUT_DIR = process.env.OPENAPI_OUT_DIR || join(ROOT, "bin/cli/api-commands"); // Operations already covered by hand-crafted commands — skip in generated output. const IGNORED_OP_IDS = new Set([ @@ -51,6 +51,29 @@ if (!existsSync(OUT_DIR)) mkdirSync(OUT_DIR, { recursive: true }); const spec = yaml.load(readFileSync(SPEC_PATH, "utf8")); +// Minimal, scoped $ref resolver — only follows refs into components/parameters. +// This is not a generic dereferencer (no cycle handling, no cross-file refs): +// OpenAPI `parameters` entries in this spec only ever $ref a component parameter +// (see docs/openapi.yaml → components/parameters/ResourceId), so a full +// dereferencer would be scope creep. Without this, `p.in === "path"` silently +// drops every $ref'd path parameter (a bare `{ $ref }` object has no `.in`), +// which is what let generated PATCH/DELETE combo commands lose --id (#10955). +const PARAM_REF_PREFIX = "#/components/parameters/"; +function resolveParam(p) { + if (p && typeof p === "object" && typeof p.$ref === "string") { + if (!p.$ref.startsWith(PARAM_REF_PREFIX)) { + throw new Error(`Unsupported parameter $ref (only ${PARAM_REF_PREFIX}* is resolved): ${p.$ref}`); + } + const name = p.$ref.slice(PARAM_REF_PREFIX.length); + const resolved = spec.components?.parameters?.[name]; + if (!resolved) { + throw new Error(`Unresolvable parameter $ref: ${p.$ref}`); + } + return resolved; + } + return p; +} + /** @type {Record>} */ const byTag = {}; @@ -89,7 +112,7 @@ for (const [tag, ops] of Object.entries(byTag)) { for (const { path, method, opId, op } of ops) { const cmdName = kebab(opId); - const params = op.parameters || []; + const params = (op.parameters || []).map(resolveParam); const pathParams = params.filter((p) => p.in === "path"); const queryParams = params.filter((p) => p.in === "query"); const hasBody = !!op.requestBody; diff --git a/scripts/dev/run-next.mjs b/scripts/dev/run-next.mjs index 325398c2861..e03cfc92903 100644 --- a/scripts/dev/run-next.mjs +++ b/scripts/dev/run-next.mjs @@ -83,8 +83,10 @@ const { dashboardPort } = runtimePorts; const hostname = process.env.HOST || "0.0.0.0"; // Turbopack by default in dev (matches the Next 16 CLI default and the production // build default in build-next-isolated.mjs); OMNIROUTE_USE_TURBOPACK=0 is the -// webpack escape hatch. -const useTurbopack = dev && mergedEnv.OMNIROUTE_USE_TURBOPACK !== "0"; +// webpack escape hatch. Under Bun, Turbopack native V8 bindings are unavailable, +// so Bun automatically disables Turbopack and uses Webpack. +const isBun = Boolean(process.versions.bun); +const useTurbopack = dev && mergedEnv.OMNIROUTE_USE_TURBOPACK !== "0" && !isBun; process.env.OMNIROUTE_WS_BRIDGE_SECRET ||= randomUUID(); // Per-process secret used to prove the trusted peer-IP stamp came from this // server (read by the authz middleware in the same process). See peer-stamp.mjs. diff --git a/scripts/dev/smoke-electron-packaged.mjs b/scripts/dev/smoke-electron-packaged.mjs index 03717912ff7..72afc2f4a7c 100644 --- a/scripts/dev/smoke-electron-packaged.mjs +++ b/scripts/dev/smoke-electron-packaged.mjs @@ -409,45 +409,115 @@ async function settleAfterReady({ getExitState, logs, settleMs }) { } } -async function main() { - const appExecutable = discoverPackagedExecutable(); - if (!existsSync(appExecutable)) { +function assertExecutableExists(appExecutable) { + if (existsSync(appExecutable)) return; + + throw new Error( + `Packaged OmniRoute executable not found at ${appExecutable}. Build it first with \`npm run build: --prefix electron\` or set ELECTRON_SMOKE_APP_EXECUTABLE.` + ); +} + +// ── CI sandbox workaround ────────────────────────────────── +// GitHub Actions runners cannot set SUID on chrome-sandbox (Linux) +// and Windows runners may fail silently without --no-sandbox. +function buildCiSpawnArgs(currentPlatform = platform()) { + if (!process.env.CI) return []; + + const spawnArgs = ["--no-sandbox", "--disable-gpu"]; + if (currentPlatform === "linux") { + spawnArgs.push("--disable-dev-shm-usage"); + } + return spawnArgs; +} + +const NATIVE_DRIVER_LOG_PATTERN = /\[DB\] Driver: (bun:sqlite|better-sqlite3|node:sqlite) \|/; +const SQLJS_DRIVER_LOG_PATTERN = /\[DB\] Driver: sql\.js \|/; + +/** + * Regression guard for #7592: on a packaged app's SECOND launch against an + * already-persisted DATA_DIR, a stale-ABI better-sqlite3 binary (resolved via + * a Turbopack-hashed import) used to fail to load and silently fall through + * to the sql.js (WASM) driver — which then OOMs/retry-loops on real-sized + * databases. Asserts the startup log shows a native driver was selected. + */ +export function assertNativeDriverSelected(logs) { + if (NATIVE_DRIVER_LOG_PATTERN.test(logs)) return; + + if (SQLJS_DRIVER_LOG_PATTERN.test(logs)) { throw new Error( - `Packaged OmniRoute executable not found at ${appExecutable}. Build it first with \`npm run build: --prefix electron\` or set ELECTRON_SMOKE_APP_EXECUTABLE.` + "Packaged Electron app fell back to the sql.js (WASM) driver instead of a native SQLite " + + "driver — this is the regression #7592 guards against (stale-ABI better-sqlite3 binary)." ); } - const smokeUrl = process.env.ELECTRON_SMOKE_URL || DEFAULT_URL; - const timeoutMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_TIMEOUT_MS, DEFAULT_TIMEOUT_MS); - const settleMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_SETTLE_MS, DEFAULT_SETTLE_MS); - const dataDir = - process.env.ELECTRON_SMOKE_DATA_DIR || - (await mkdtemp(join(tmpdir(), "omniroute-electron-smoke-"))); - const removeDataDir = - !process.env.ELECTRON_SMOKE_DATA_DIR && process.env.ELECTRON_SMOKE_KEEP_DATA !== "1"; - const smokeEnv = buildSmokeEnv({ dataDir }); + throw new Error( + "Packaged Electron app logs contain no '[DB] Driver: ...' line — cannot confirm which SQLite " + + "driver loaded." + ); +} - await assertPortIsFree(smokeUrl); - await ensureSmokeEnvDirs(smokeEnv, dataDir); +async function waitForReady({ logs, smokeUrl, timeoutMs, settleMs, exitState }) { + const startedAt = Date.now(); + let lastError = null; - // ── CI sandbox workaround ────────────────────────────────── - // GitHub Actions runners cannot set SUID on chrome-sandbox (Linux) - // and Windows runners may fail silently without --no-sandbox. - const spawnArgs = []; - if (process.env.CI) { - spawnArgs.push("--no-sandbox", "--disable-gpu"); - if (platform() === "linux") { - spawnArgs.push("--disable-dev-shm-usage"); + while (Date.now() - startedAt < timeoutMs) { + assertNoFatalLogs(logs.value); + + if (exitState.spawnError !== null) { + throw new Error(`Packaged Electron app failed to launch: ${exitState.spawnError.message}`); } + if (exitState.exitCode !== null || exitState.signalCode !== null) { + throw new Error( + `Packaged Electron app exited before readiness: code=${exitState.exitCode} signal=${exitState.signalCode}` + ); + } + + try { + const response = await fetchWithTimeout(smokeUrl, 1_000); + if (response.status === 200) { + assertNoFatalLogs(logs.value); + console.log(`[electron-smoke] ready: ${smokeUrl} returned HTTP 200`); + await settleAfterReady({ + getExitState: () => ({ exitCode: exitState.exitCode, signalCode: exitState.signalCode }), + logs, + settleMs, + }); + console.log(`[electron-smoke] stable for ${settleMs}ms after readiness`); + return; + } + lastError = new Error(`HTTP ${response.status}`); + } catch (error) { + lastError = error; + } + + await sleep(500); } + throw new Error( + `Packaged Electron app did not serve ${smokeUrl} within ${timeoutMs}ms. Last error: ${ + lastError instanceof Error ? lastError.message : String(lastError) + }` + ); +} + +/** + * Launches the packaged app once against `dataDir`, waits for readiness + + * settle, tears it down, and returns the captured stdout/stderr text. Shared + * by the single-launch path and the cold-restart (two-launch) path so both + * exercise identical spawn/readiness/shutdown behavior. + */ +async function launchAndCollectLogs({ appExecutable, smokeUrl, dataDir, timeoutMs, settleMs, streamLogs }) { + const smokeEnv = buildSmokeEnv({ dataDir }); + await assertPortIsFree(smokeUrl); + await ensureSmokeEnvDirs(smokeEnv, dataDir); + + const spawnArgs = buildCiSpawnArgs(); console.log(`[electron-smoke] launching ${appExecutable}`); if (spawnArgs.length) console.log(`[electron-smoke] CI args: ${spawnArgs.join(" ")}`); console.log(`[electron-smoke] DATA_DIR=${dataDir}`); console.log(`[electron-smoke] waiting for ${smokeUrl}`); const logs = { value: "" }; - const streamLogs = process.env.ELECTRON_SMOKE_STREAM_LOGS === "1"; const child = spawn(appExecutable, spawnArgs, { detached: platform() !== "win32", env: smokeEnv, @@ -457,60 +527,18 @@ async function main() { child.stdout?.on("data", (chunk) => appendLog(logs, chunk, "[electron] ", streamLogs)); child.stderr?.on("data", (chunk) => appendLog(logs, chunk, "[electron:err] ", streamLogs)); - let exitCode = null; - let signalCode = null; - let spawnError = null; + const exitState = { exitCode: null, signalCode: null, spawnError: null }; child.once("exit", (code, signal) => { - exitCode = code; - signalCode = signal; + exitState.exitCode = code; + exitState.signalCode = signal; }); child.once("error", (error) => { - spawnError = error; + exitState.spawnError = error; }); try { - const startedAt = Date.now(); - let lastError = null; - - while (Date.now() - startedAt < timeoutMs) { - assertNoFatalLogs(logs.value); - - if (spawnError !== null) { - throw new Error(`Packaged Electron app failed to launch: ${spawnError.message}`); - } - - if (exitCode !== null || signalCode !== null) { - throw new Error( - `Packaged Electron app exited before readiness: code=${exitCode} signal=${signalCode}` - ); - } - - try { - const response = await fetchWithTimeout(smokeUrl, 1_000); - if (response.status === 200) { - assertNoFatalLogs(logs.value); - console.log(`[electron-smoke] ready: ${smokeUrl} returned HTTP 200`); - await settleAfterReady({ - getExitState: () => ({ exitCode, signalCode }), - logs, - settleMs, - }); - console.log(`[electron-smoke] stable for ${settleMs}ms after readiness`); - return; - } - lastError = new Error(`HTTP ${response.status}`); - } catch (error) { - lastError = error; - } - - await new Promise((resolve) => setTimeout(resolve, 500)); - } - - throw new Error( - `Packaged Electron app did not serve ${smokeUrl} within ${timeoutMs}ms. Last error: ${ - lastError instanceof Error ? lastError.message : String(lastError) - }` - ); + await waitForReady({ logs, smokeUrl, timeoutMs, settleMs, exitState }); + return logs.value; } catch (error) { if (!streamLogs) { printLogTail(logs.value); @@ -519,6 +547,43 @@ async function main() { } finally { await stopApp(child); await waitForPortClosed(smokeUrl); + } +} + +async function main() { + const appExecutable = discoverPackagedExecutable(); + assertExecutableExists(appExecutable); + + const smokeUrl = process.env.ELECTRON_SMOKE_URL || DEFAULT_URL; + const timeoutMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_TIMEOUT_MS, DEFAULT_TIMEOUT_MS); + const settleMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_SETTLE_MS, DEFAULT_SETTLE_MS); + const streamLogs = process.env.ELECTRON_SMOKE_STREAM_LOGS === "1"; + // #7592: rerun against the SAME (persisted) DATA_DIR and assert the second + // launch selected a native SQLite driver, not the sql.js WASM fallback. + const coldRestart = process.env.ELECTRON_SMOKE_COLD_RESTART === "1"; + const dataDir = + process.env.ELECTRON_SMOKE_DATA_DIR || + (await mkdtemp(join(tmpdir(), "omniroute-electron-smoke-"))); + const removeDataDir = + !process.env.ELECTRON_SMOKE_DATA_DIR && process.env.ELECTRON_SMOKE_KEEP_DATA !== "1"; + + try { + await launchAndCollectLogs({ appExecutable, smokeUrl, dataDir, timeoutMs, settleMs, streamLogs }); + + if (!coldRestart) return; + + console.log("[electron-smoke] cold-restart: relaunching against the same DATA_DIR"); + const secondLaunchLogs = await launchAndCollectLogs({ + appExecutable, + smokeUrl, + dataDir, + timeoutMs, + settleMs, + streamLogs, + }); + assertNativeDriverSelected(secondLaunchLogs); + console.log("[electron-smoke] cold-restart: native SQLite driver confirmed on second launch"); + } finally { if (removeDataDir) { await rm(dataDir, { recursive: true, force: true }); } diff --git a/skills/cli-routing/SKILL.md b/skills/cli-routing/SKILL.md index 9485d0a5930..aa7591e402a 100644 --- a/skills/cli-routing/SKILL.md +++ b/skills/cli-routing/SKILL.md @@ -70,6 +70,11 @@ omniroute combo switch Create a new routing combo +**Flags:** + +- `--models ` +- `--model ` + **Example:** ```bash diff --git a/skills/omni-combos-routing/SKILL.md b/skills/omni-combos-routing/SKILL.md index 549e2b75e54..9afb31e36d7 100644 --- a/skills/omni-combos-routing/SKILL.md +++ b/skills/omni-combos-routing/SKILL.md @@ -60,6 +60,8 @@ curl -X PUT https://localhost:20128/api/combos/{id} \ Update combo +Partial update: the body is merged onto the stored combo, so a field left out keeps its current value. An array that IS sent replaces the stored one outright. + ```bash curl -X PATCH https://localhost:20128/api/combos/{id} \ -H "Authorization: Bearer $OMNIROUTE_TOKEN" \ diff --git a/skills/omni-webhooks/SKILL.md b/skills/omni-webhooks/SKILL.md index 251b5f58176..c60df46ecab 100644 --- a/skills/omni-webhooks/SKILL.md +++ b/skills/omni-webhooks/SKILL.md @@ -1,12 +1,12 @@ --- name: omni-webhooks -description: Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries. +description: Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries. --- ## Overview -Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries. +Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries. ## Authentication diff --git a/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx new file mode 100644 index 00000000000..77cf1932983 --- /dev/null +++ b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx @@ -0,0 +1,108 @@ +"use client"; + +import { useSyncExternalStore } from "react"; +import { useTranslations } from "next-intl"; +import ProviderIcon from "@/shared/components/ProviderIcon"; + +// Branded short link through our own link.omniroute.online shortener, so the +// click lands in our Kutt metrics. Points at cheaperinference.com?utm_source=omniroute +// (the URL in README.md's Open Source Friends section). Keep in sync with the +// `cheaper` slug on the shortener. +const CHEAPER_INFERENCE_URL = "https://link.omniroute.online/cheaper"; + +// Cheaper Inference brand green (#31f889). White text on it fails contrast, so +// the CTA pairs it with the dark ink from the provider's color token (colors.ts: +// cheaperinference.text = #04170d). Hex values stay in sync with that token. + +const DISMISS_STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +// Same-tab signal for the dismiss button, since writing localStorage doesn't +// fire a "storage" event in the tab that wrote it. +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; + +function isNotDismissed(): boolean { + try { + return !localStorage.getItem(DISMISS_STORAGE_KEY); + } catch { + return true; + } +} + +function subscribe(callback: () => void) { + window.addEventListener(DISMISS_EVENT, callback); + return () => window.removeEventListener(DISMISS_EVENT, callback); +} + +// SSR has no localStorage, so the server always renders the banner visible; +// useSyncExternalStore reconciles that against the real client-side value +// right after hydration, mirroring KimiSponsorBanner's pattern. +function getServerSnapshot() { + return true; +} + +/** + * Dismissable banner announcing the Cheaper Inference OmniRoute partnership on + * the dashboard home page — same size/shape as KimiSponsorBanner, no version + * gate (durable partnership, not a time-boxed offer). The logomark reuses + * . + */ +export default function CheaperInferenceSponsorBanner() { + const t = useTranslations("cheaperInferenceSponsorBanner"); + const visible = useSyncExternalStore(subscribe, isNotDismissed, getServerSnapshot); + + if (!visible) { + return null; + } + + const dismiss = () => { + try { + localStorage.setItem(DISMISS_STORAGE_KEY, "true"); + } catch { + // ignore — worst case the banner reappears next visit + } + window.dispatchEvent(new Event(DISMISS_EVENT)); + }; + + return ( +
+
+
+ +
+
+

{t("title")}

+

{t("description")}

+
+
+ +
+
+ + {t("cta")} + + + {t("partnerLinkNote")} +
+ +
+
+ ); +} diff --git a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx index 0b23976531c..1715f17c839 100644 --- a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx +++ b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx @@ -6,7 +6,9 @@ import { useTranslations } from "next-intl"; // Marketplace listing is the primary CTA; Open VSX (Cursor/Windsurf/VSCodium/etc.) // is called out via secondaryNote instead of a second button, to keep this banner // the same size as KimiSponsorBanner. -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +// Branded short link through our own link.omniroute.online shortener (the `vsx` +// slug), so the click lands in our Kutt metrics. +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; const DISMISS_STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; // Same-tab signal for the dismiss button, since writing localStorage doesn't diff --git a/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx b/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx index 7970c40a8ee..8d8c5c55738 100644 --- a/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx +++ b/src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx @@ -291,7 +291,9 @@ function ComboAutopilotPanel({ report }: { report: ComboAutopilotReport }) { icon="monitor_heart" label={t("comboHealthIssues")} value={report.summary.issueCount.toLocaleString()} - subValue={t("comboHealthActionable", { count: report.summary.actionableCount })} + subValue={t("comboHealthActionable", { + count: report.summary.suggestionCount ?? report.summary.actionableCount ?? 0, + })} /> +) { + if (typeof t?.has === "function" && t.has(key)) return t(key, values); return fallback; } @@ -94,10 +99,9 @@ export default function IntelligentComboPanel({ const updatedCombo = await response.json(); onComboUpdated?.(updatedCombo); notify.success( - getI18nOrFallback(t, "modePackUpdated", "Mode pack updated to {pack}.").replace( - "{pack}", - modePackId - ) + getI18nOrFallback(t, "modePackUpdated", "Mode pack updated to {pack}.", { + pack: modePackId, + }).replace("{pack}", modePackId) ); } catch (error: any) { notify.error(error?.message || "Failed to update mode pack."); @@ -184,10 +188,9 @@ export default function IntelligentComboPanel({ {savingModePack && ( - {getI18nOrFallback(t, "savingModePack", "Saving {pack}…").replace( - "{pack}", - savingModePack - )} + {getI18nOrFallback(t, "savingModePack", "Saving {pack}…", { + pack: savingModePack, + }).replace("{pack}", savingModePack)} )} diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index fceada4eb56..7cb1c6c0eea 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -533,9 +533,9 @@ function getStrategyBadgeClass(strategy) { return "bg-blue-500/15 text-blue-600 dark:text-blue-400"; } -function getI18nOrFallback(t, key, fallback) { +function getI18nOrFallback(t, key, fallback, values) { try { - if (typeof t.has === "function" && t.has(key)) return t(key); + if (typeof t.has === "function" && t.has(key)) return t(key, values); } catch {} return fallback; } @@ -1565,7 +1565,8 @@ function StrategyRecommendationsPanel({ strategy, onApply, showNudge }) { {getI18nOrFallback( t, "recommendationsUpdated", - "Recommendations updated for {strategy}." + "Recommendations updated for {strategy}.", + { strategy: strategyLabel } ).replace("{strategy}", strategyLabel)} )} diff --git a/src/app/(dashboard)/dashboard/onboarding/page.tsx b/src/app/(dashboard)/dashboard/onboarding/page.tsx index f0164d63acb..0d9f5376e45 100644 --- a/src/app/(dashboard)/dashboard/onboarding/page.tsx +++ b/src/app/(dashboard)/dashboard/onboarding/page.tsx @@ -318,6 +318,11 @@ export default function OnboardingWizard() { /> {t("skipPassword")} + {skipSecurity && ( +

+ {t("securityDescSkipWarning")} +

+ )} {!skipSecurity && (

{t("providerDesc")}

- -
- - {t("freeProviders.orUseApiKey")} - -
-
- {COMMON_PROVIDERS.map((p) => ( - - ))} -
- {selectedProvider && ( + {skipSecurity && ( +
+

{t("providerRequiresPassword")}

+
+ )} + {!skipSecurity && } + {!skipSecurity && ( +
+ + {t("freeProviders.orUseApiKey")} + +
+ )} + {!skipSecurity && ( +
+ {COMMON_PROVIDERS.map((p) => ( + + ))} +
+ )} + {!skipSecurity && selectedProvider && (
)} - {currentStep.id === "provider" && ( + {currentStep.id === "provider" && !skipSecurity ? ( - )} + ) : null} {currentStep.id === "test" && ( + + {providerText( + t, + "harImportButtonHint", + "Export from DevTools Network tab after sending at least one chat message." + )} + + { + void handleFile(event.target.files?.[0]); + event.target.value = ""; + }} + /> +
+ {state.phase === "error" && ( +

+ {state.message} +

+ )} + {state.phase === "success" && expiryText && ( +

+ {expiryText} +

+ )} +
+ ); +} diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx index 8bdddc2aa73..9a2bdc00380 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx @@ -31,6 +31,7 @@ import { import { getWebSessionCredentialRequirement } from "../../webSessionCredentials"; import { useOpenRouterPresetControl } from "../OpenRouterPresetInput"; import WebSessionCredentialGuide from "../WebSessionCredentialGuide"; +import HarImportButton from "../HarImportButton"; import CcCompatibleRequestDefaultsFields from "./CcCompatibleRequestDefaultsFields"; import { buildAddProviderSpecificData } from "./connectionProviderSpecificData"; import { getCommandCodeAuthPhaseLabel } from "./commandCodeAuthPhase"; @@ -155,7 +156,7 @@ export default function AddApiKeyModal({ if (!isOpen || wasOpen) return; // On open, reset baseUrl and assign a unique default name so a second API key // for the same provider doesn't reuse "main" and trigger the backend - // name-based upsert that would silently overwrite the first connection (#6499). + // name-based upsert that would silently overwrite the first connection (#6499, #11033). setFormData((current) => ({ ...current, name: computeConnectionDefaultName(existingConnectionCount), @@ -209,13 +210,13 @@ export default function AddApiKeyModal({ ? "Freebuff uses an authentic CLI auth token obtained via codebuff CLI login or automated harvester." : isWebSessionCredential ? getWebSessionCredentialHint(t, webSessionCredential, providerDisplayName, false) - : isLocalSelfHostedProvider - ? t("localProviderApiKeyOptionalHint", { - provider: localProviderMetadata?.name || providerName || provider || "", - }) - : apiKeyOptional - ? t("apiKeyOptionalHint") - : undefined; + : isLocalSelfHostedProvider + ? t("localProviderApiKeyOptionalHint", { + provider: localProviderMetadata?.name || providerName || provider || "", + }) + : apiKeyOptional + ? t("apiKeyOptionalHint") + : undefined; const credentialValidationFailedMessage = isWebSessionCredential ? providerText( t, @@ -750,40 +751,54 @@ export default function AddApiKeyModal({ t={t} /> )} - {!isNoAuthWebSessionCredential && ( -
- setFormData({ ...formData, apiKey: e.target.value })} - className="flex-1" - placeholder={apiCredentialPlaceholder} - hint={apiCredentialHint} - autoComplete="off" - spellCheck={false} - autoCapitalize="off" - /> -
- -
-
+ {provider && ( + setFormData({ ...formData, apiKey })} + /> )} + {!isNoAuthWebSessionCredential && (() => { + const isCheckDisabled = + (!isCompatible && !apiKeyOptional && !formData.apiKey) || + (isGooglePse && !formData.cx.trim()) || + validating || + saving; + return ( +
+ setFormData({ ...formData, apiKey: e.target.value })} + onKeyDown={(e) => { + if (e.key === "Enter" && !isCheckDisabled) { + e.preventDefault(); + handleValidate(); + } + }} + className="flex-1" + placeholder={apiCredentialPlaceholder} + hint={apiCredentialHint} + autoComplete="off" + spellCheck={false} + autoCapitalize="off" + /> +
+ +
+
+ ); + })()} {isChatGptWebCodex && (
diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx index 39028de1367..82c41e82c44 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx @@ -49,6 +49,7 @@ import { import { getWebSessionCredentialRequirement } from "../../webSessionCredentials"; import { useOpenRouterPresetControl } from "../OpenRouterPresetInput"; import WebSessionCredentialGuide from "../WebSessionCredentialGuide"; +import HarImportButton from "../HarImportButton"; import CcCompatibleRequestDefaultsFields from "./CcCompatibleRequestDefaultsFields"; import { CodexConnectionFields } from "./CodexFingerprintFields"; import { assignEditApiKeyProviderSpecificData } from "./connectionProviderSpecificData"; @@ -909,6 +910,12 @@ export default function EditConnectionModal({ t={t} /> )} + {provider && ( + setFormData({ ...formData, apiKey })} + /> + )} {!isNoAuthWebSessionCredential && (
(typeof item === "string" ? item : item?.name ?? "")) + .filter(Boolean) + ); + if (!names.has("main")) return "main"; + let index = 2; + while (names.has(`main-${index}`)) { + index++; + } + return `main-${index}`; + } + + const count = existingConnectionCountOrConnections ?? 0; return count <= 0 ? "main" : `main-${count + 1}`; } diff --git a/src/app/(dashboard)/home/page.tsx b/src/app/(dashboard)/home/page.tsx index beccde22610..bc10df88f42 100644 --- a/src/app/(dashboard)/home/page.tsx +++ b/src/app/(dashboard)/home/page.tsx @@ -4,6 +4,7 @@ import { getSettings } from "@/lib/localDb"; import HomePageClient from "../dashboard/HomePageClient"; import BootstrapBanner from "../dashboard/BootstrapBanner"; import KimiSponsorBanner from "../dashboard/KimiSponsorBanner"; +import CheaperInferenceSponsorBanner from "../dashboard/CheaperInferenceSponsorBanner"; import VscodeCopilotBanner from "../dashboard/VscodeCopilotBanner"; import NewsBanner from "../dashboard/NewsBanner"; @@ -20,6 +21,7 @@ export default async function HomePage() { <> {isBootstrapped && } + diff --git a/src/app/a2a/route.ts b/src/app/a2a/route.ts index e9e90a99abb..af7d93a2e50 100644 --- a/src/app/a2a/route.ts +++ b/src/app/a2a/route.ts @@ -17,6 +17,8 @@ import { logRoutingDecision } from "@/lib/a2a/routingLogger"; import { createA2AStream, SSE_HEADERS } from "@/lib/a2a/streaming"; import { A2A_SKILL_HANDLERS, executeA2ATaskWithState } from "@/lib/a2a/taskExecution"; import { getSettings } from "@/lib/db/settings"; +import { isRequireApiKeyEnabled } from "@/shared/utils/featureFlags"; +import { extractApiKey, isValidApiKey } from "@/sse/services/auth"; // ============ A2A v1.0 ↔ v0.3 compatibility layer ============ // A2A 1.0 renamed the JSON-RPC methods (message/send → SendMessage, @@ -136,14 +138,25 @@ function tokensMatch(provided: string, expected: string): boolean { return timingSafeEqual(a, b); } -function authenticate(req: NextRequest): boolean { - // If no API key is configured, allow all requests +async function authenticate(req: NextRequest): Promise { + // /a2a is outside the authz proxy matcher, so the REQUIRE_API_KEY posture the + // pipeline enforces for /v1 never ran here — the route accepted every caller + // whenever OMNIROUTE_API_KEY was unset, which is the shipped default + // (GHSA-v54m-6rm3-p565). Apply the same posture directly: when a client key is + // required, demand a valid OmniRoute key; otherwise honor the legacy explicit + // A2A key; otherwise stay keyless (the same local-first default as /v1). + const apiKey = extractApiKey(req); + if (isRequireApiKeyEnabled()) { + return apiKey ? await isValidApiKey(apiKey) : false; + } + const configuredKey = process.env.OMNIROUTE_API_KEY; - if (!configuredKey) return true; + if (configuredKey) { + return apiKey ? tokensMatch(apiKey, configuredKey) : false; + } - const authHeader = req.headers.get("authorization") || ""; - const token = authHeader.replace(/^Bearer\s+/i, ""); - return tokensMatch(token, configuredKey); + // No API key required and none configured — allow (keyless local-first). + return true; } // ============ JSON-RPC Helpers ============ @@ -179,7 +192,7 @@ async function rejectIfA2ADisabled(id: string | number | null) { export async function POST(req: NextRequest) { // Auth check - if (!authenticate(req)) { + if (!(await authenticate(req))) { return jsonRpcError(null, -32600, "Unauthorized: missing or invalid API key"); } diff --git a/src/app/api/acp/agents/route.ts b/src/app/api/acp/agents/route.ts index 9f8fc6bf829..1aa6c1a3305 100644 --- a/src/app/api/acp/agents/route.ts +++ b/src/app/api/acp/agents/route.ts @@ -1,5 +1,7 @@ import { NextResponse } from "next/server"; import { z } from "zod"; + +export const dynamic = "force-dynamic"; import { type CliAgentInfo, detectInstalledAgents, diff --git a/src/app/api/admin/concurrency/route.ts b/src/app/api/admin/concurrency/route.ts index 0d6f59bf51f..550812e91b9 100644 --- a/src/app/api/admin/concurrency/route.ts +++ b/src/app/api/admin/concurrency/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { getAllRateLimitStatus } from "@omniroute/open-sse/services/rateLimitManager.ts"; import { getStats as getSemaphoreStats, diff --git a/src/app/api/analytics/compression/route.ts b/src/app/api/analytics/compression/route.ts index a81e6fef804..7006cac6eba 100644 --- a/src/app/api/analytics/compression/route.ts +++ b/src/app/api/analytics/compression/route.ts @@ -6,6 +6,7 @@ */ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { getCompressionAnalyticsSummary } from "@/lib/db/compressionAnalytics"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; diff --git a/src/app/api/auth/csrf/route.ts b/src/app/api/auth/csrf/route.ts index 88fae6c3a26..a18026b4474 100644 --- a/src/app/api/auth/csrf/route.ts +++ b/src/app/api/auth/csrf/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { issueDashboardCsrfToken } from "@/server/authz/csrf"; diff --git a/src/app/api/auth/status/route.ts b/src/app/api/auth/status/route.ts index 9cc4d872c6e..62cb1a71918 100644 --- a/src/app/api/auth/status/route.ts +++ b/src/app/api/auth/status/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { cookies } from "next/headers"; import { jwtVerify } from "jose"; diff --git a/src/app/api/batches/[id]/route.ts b/src/app/api/batches/[id]/route.ts index 83f23c7e0d3..5d4a73f3d2d 100644 --- a/src/app/api/batches/[id]/route.ts +++ b/src/app/api/batches/[id]/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { getBatch } from "@/lib/localDb"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; diff --git a/src/app/api/batches/route.ts b/src/app/api/batches/route.ts index e67ff13507f..4c8c54d46e0 100644 --- a/src/app/api/batches/route.ts +++ b/src/app/api/batches/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { listBatches } from "@/lib/localDb"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; diff --git a/src/app/api/cache/entries/route.ts b/src/app/api/cache/entries/route.ts index d2a7ab4bc7e..aa2e60b04cf 100644 --- a/src/app/api/cache/entries/route.ts +++ b/src/app/api/cache/entries/route.ts @@ -1,4 +1,5 @@ import { NextRequest, NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { isAuthenticated } from "@/shared/utils/apiAuth"; import { listSemanticCacheEntries, diff --git a/src/app/api/cache/reasoning/route.ts b/src/app/api/cache/reasoning/route.ts index 89769a2dcf2..cfa6805c4dc 100644 --- a/src/app/api/cache/reasoning/route.ts +++ b/src/app/api/cache/reasoning/route.ts @@ -1,4 +1,5 @@ import { NextRequest, NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { isAuthenticated } from "@/shared/utils/apiAuth"; import { clearReasoningCacheAll, diff --git a/src/app/api/cache/route.ts b/src/app/api/cache/route.ts index 01d310432cd..4fcb5bdec8e 100644 --- a/src/app/api/cache/route.ts +++ b/src/app/api/cache/route.ts @@ -1,4 +1,5 @@ import { NextRequest, NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { getCacheStats, clearCache, diff --git a/src/app/api/cache/stats/route.ts b/src/app/api/cache/stats/route.ts index 33f3c447f9b..43e48b5e9cc 100644 --- a/src/app/api/cache/stats/route.ts +++ b/src/app/api/cache/stats/route.ts @@ -1,4 +1,5 @@ import { NextRequest, NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { clearMemoryCache, getMemoryCacheStats } from "@/lib/semanticCache"; import { isAuthenticated } from "@/shared/utils/apiAuth"; import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; diff --git a/src/app/api/free-provider-rankings/route.ts b/src/app/api/free-provider-rankings/route.ts index e493e179a0d..96bbd53f8c5 100644 --- a/src/app/api/free-provider-rankings/route.ts +++ b/src/app/api/free-provider-rankings/route.ts @@ -22,6 +22,11 @@ const QuerySchema = z.object({ // Additive filters (default off → current behavior). `availableOnly` implies configured. configuredOnly: boolParam, availableOnly: boolParam, + // Opt-in usage reporting: costs one aggregate query, so it is never implicit. + withUsage: boolParam, + // Rejected rather than silently coerced: a typo must not quietly return a + // different window than the caller asked for. + usageRange: z.enum(["1h", "24h", "7d", "30d"]).optional(), }); export async function OPTIONS() { @@ -35,6 +40,8 @@ export async function GET(request: NextRequest) { limit: url.searchParams.get("limit") || undefined, configuredOnly: url.searchParams.get("configuredOnly") || undefined, availableOnly: url.searchParams.get("availableOnly") || undefined, + withUsage: url.searchParams.get("withUsage") || undefined, + usageRange: url.searchParams.get("usageRange") || undefined, }); if (!parsed.success) { @@ -44,10 +51,12 @@ export async function GET(request: NextRequest) { ); } - const { category, limit, configuredOnly, availableOnly } = parsed.data; + const { category, limit, configuredOnly, availableOnly, withUsage, usageRange } = parsed.data; const rankings = await computeFreeProviderRankings(category, limit, { configuredOnly, availableOnly, + withUsage, + usageRange, }); return NextResponse.json({ rankings }, { headers: CORS_HEADERS }); diff --git a/src/app/api/gamification/federation/leaderboard/route.ts b/src/app/api/gamification/federation/leaderboard/route.ts index e198ae2aef8..26e28461223 100644 --- a/src/app/api/gamification/federation/leaderboard/route.ts +++ b/src/app/api/gamification/federation/leaderboard/route.ts @@ -38,7 +38,14 @@ export async function GET(request: NextRequest) { const url = new URL(request.url); const scope: LeaderboardScope = (url.searchParams.get("scope") || "global") as LeaderboardScope; - const limit = Number(url.searchParams.get("limit") || 100); + const rawLimit = url.searchParams.get("limit"); + const limit = rawLimit === null ? 100 : Number(rawLimit); + if (!Number.isInteger(limit) || limit < 1 || limit > 200) { + return NextResponse.json( + { error: "'limit' must be an integer between 1 and 200" }, + { status: 400, headers: CORS_HEADERS } + ); + } const entries = await getTopN(scope, limit); diff --git a/src/app/api/gamification/leaderboard/route.ts b/src/app/api/gamification/leaderboard/route.ts index 33159af39cd..cded801d13a 100644 --- a/src/app/api/gamification/leaderboard/route.ts +++ b/src/app/api/gamification/leaderboard/route.ts @@ -18,10 +18,18 @@ export async function GET(request: NextRequest) { const url = new URL(request.url); const scope = (url.searchParams.get("scope") || "global") as LeaderboardScope; - const limit = Number(url.searchParams.get("limit") || 50); + const rawLimit = url.searchParams.get("limit"); + const limit = rawLimit === null ? 50 : Number(rawLimit); const apiKeyId = url.searchParams.get("apiKeyId"); - const entries = await getTopN(scope, Math.min(limit, 200)); + if (!Number.isInteger(limit) || limit < 1 || limit > 200) { + return NextResponse.json( + { error: "'limit' must be an integer between 1 and 200" }, + { status: 400, headers: CORS_HEADERS } + ); + } + + const entries = await getTopN(scope, limit); let myRank: number | null = null; let neighbors = null; diff --git a/src/app/api/local/redis/status/route.ts b/src/app/api/local/redis/status/route.ts index 7f7cb47bc57..0ab37e76fdc 100644 --- a/src/app/api/local/redis/status/route.ts +++ b/src/app/api/local/redis/status/route.ts @@ -63,21 +63,51 @@ async function pingRedis(port: string): Promise { }); } +function parseRedisUrl(url?: string): { host: string; port: number } | null { + if (!url) return null; + try { + const u = new URL(url); + return { host: u.hostname || "127.0.0.1", port: Number(u.port) || 6379 }; + } catch { + return null; + } +} + export async function GET() { const guard = isLocalRequestAllowed(); if (!guard.allowed) { - return NextResponse.json({ error: guard.reason }, { status: 403 }); + const reason = (guard as { reason?: string }).reason ?? "Forbidden: not a loopback request"; + return NextResponse.json({ error: reason }, { status: 403 }); } + // Docker/Podman container state (the 1-click launcher path). const runtime = await detectRuntime(); - if (!runtime) { - return NextResponse.json( - { exists: false, running: false, reachable: false, error: "No container runtime (podman or docker) found on PATH" }, - { status: 503 } - ); + let container = { exists: false, running: false, reachable: false }; + if (runtime) { + const { exists, running } = await containerState(runtime); + const reachable = running ? await pingRedis(HOST_PORT) : false; + container = { exists, running, reachable }; } - const { exists, running } = await containerState(runtime); - const reachable = running ? await pingRedis(HOST_PORT) : false; - return NextResponse.json({ runtime, name: CONTAINER_NAME, port: HOST_PORT, exists, running, reachable }); + // Native Redis via REDIS_URL (the production path this instance uses). OmniRoute + // is "connected" whenever REDIS_URL is configured AND the server answers — even + // when no Docker container is present. + const redisUrl = process.env.REDIS_URL?.trim() || ""; + const parsed = parseRedisUrl(redisUrl); + const redisUrlReachable = parsed ? await pingRedis(String(parsed.port)) : false; + + const running = container.running || redisUrlReachable; + const reachable = container.reachable || redisUrlReachable; + const exists = container.exists || redisUrlReachable; + + return NextResponse.json({ + runtime: runtime ?? null, + name: CONTAINER_NAME, + port: HOST_PORT, + exists, + running, + reachable, + redisUrlConfigured: Boolean(redisUrl), + redisUrlReachable, + }); } \ No newline at end of file diff --git a/src/app/api/monitoring/health/route.ts b/src/app/api/monitoring/health/route.ts index 144d0a3ce06..f3dceac5583 100644 --- a/src/app/api/monitoring/health/route.ts +++ b/src/app/api/monitoring/health/route.ts @@ -5,6 +5,7 @@ import { readRunningBuildSha } from "@/lib/monitoring/buildSha"; import { APP_CONFIG } from "@/shared/constants/config"; import { AI_PROVIDERS } from "@/shared/constants/providers"; import { isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; /** * GET /api/monitoring/health — System health overview @@ -20,10 +21,25 @@ import { isAuthenticated } from "@/shared/utils/apiAuth"; let healthPayloadCache: { payload: unknown; expiresAt: number } | null = null; const HEALTH_PAYLOAD_TTL_MS = 1000; -export async function GET() { +// GHSA-mvf8-qc78-5mxm: the full health payload fingerprints the host (version, +// node version, pid, memory, provider config). An anonymous caller — the common +// case on a keyless install, and what a liveness/load-balancer probe needs — gets +// only the liveness verdict; the detail is reserved for a management principal. +function publicHealthView(payload: unknown): Record { + const p = (payload ?? {}) as Record; + return { + status: p.status ?? "unknown", + ...(p.setupComplete !== undefined ? { setupComplete: p.setupComplete } : {}), + }; +} + +export async function GET(request: Request) { + const fullView = (await requireManagementAuth(request, { alwaysRequireAuth: true })) === null; const cachedNow = Date.now(); if (healthPayloadCache && cachedNow <= healthPayloadCache.expiresAt) { - return NextResponse.json(healthPayloadCache.payload); + return NextResponse.json( + fullView ? healthPayloadCache.payload : publicHealthView(healthPayloadCache.payload) + ); } const readHealthValue = (label: string, reader: () => T, fallback: T): T => { @@ -187,7 +203,7 @@ export async function GET() { }); healthPayloadCache = { payload, expiresAt: Date.now() + HEALTH_PAYLOAD_TTL_MS }; - return NextResponse.json(payload); + return NextResponse.json(fullView ? payload : publicHealthView(payload)); } catch (error) { console.error("[API] GET /api/monitoring/health error:", error); return NextResponse.json({ diff --git a/src/app/api/oauth/[provider]/[action]/route.ts b/src/app/api/oauth/[provider]/[action]/route.ts index 7b71a07758c..b52bae42012 100755 --- a/src/app/api/oauth/[provider]/[action]/route.ts +++ b/src/app/api/oauth/[provider]/[action]/route.ts @@ -24,6 +24,7 @@ import { } from "@/models"; import { getConsistentMachineId } from "@/shared/utils/machineId"; import { isValidGheUrl } from "@/shared/validation/providerSpecificData"; +import { AWS_REGION_PATTERN } from "@/lib/oauth/constants/oauth"; import { syncToCloud } from "@/lib/cloudSync"; import { startLocalServer } from "@/lib/oauth/utils/server"; import { runWithProxyContextOrDirect } from "@omniroute/open-sse/utils/proxyFetch.ts"; @@ -221,6 +222,16 @@ export async function GET( (requestDeviceCode as any)(provider, null, providerOverrideConfig) ); } else if ((provider === "kiro" || provider === "amazon-q") && startUrl) { + // GHSA-7x63: `region` is interpolated into the AWS OIDC endpoint URLs + // below, which requestDeviceCode() then fetches. Validate it against the + // canonical AWS region shape before it can steer the outbound host to an + // attacker-chosen target (userinfo/fragment tricks → SSRF / metadata). + if (!AWS_REGION_PATTERN.test(region)) { + return NextResponse.json( + { error: "region must be a valid AWS region (e.g. us-east-1)" }, + { status: 400 } + ); + } const providerOverrideConfig = { ...providerData.config, startUrl, diff --git a/src/app/api/oauth/cliproxy-import/route.ts b/src/app/api/oauth/cliproxy-import/route.ts index 7c60ba97756..ca579983b8f 100644 --- a/src/app/api/oauth/cliproxy-import/route.ts +++ b/src/app/api/oauth/cliproxy-import/route.ts @@ -3,7 +3,7 @@ import path from "path"; import { NextResponse } from "next/server"; import { createProviderConnection } from "@/models"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error"; import { scanCliProxyAuthDir, @@ -23,9 +23,9 @@ function cliProxyConfigDir(): string { } async function requireImportAuth(request: Request) { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } export async function GET(request: Request) { diff --git a/src/app/api/oauth/codex/import-token/route.ts b/src/app/api/oauth/codex/import-token/route.ts index 4601b2ae2d4..a5b22c4e37d 100644 --- a/src/app/api/oauth/codex/import-token/route.ts +++ b/src/app/api/oauth/codex/import-token/route.ts @@ -3,7 +3,7 @@ import { z } from "zod"; import { extractCodexAccountInfo } from "@/lib/oauth/services/codexImport"; import { parseCodexSessionJson } from "@/lib/oauth/utils/codexSessionImport"; import { createProviderConnection } from "@/models"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { buildErrorBody, sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; /** @@ -93,10 +93,11 @@ async function parseRequestBody( return { ok: true, resolved: resolved.resolved }; } -async function requireAuth(request: Request): Promise { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json(buildErrorBody(401, "Unauthorized"), { status: 401 }); +async function requireAuth(request: Request): Promise { + // GHSA-mg76: importing a provider connection is a state-mutating admin action. + // Require management scope (or a dashboard session) rather than accepting any + // valid client key, which the PUBLIC /api/oauth/ classification otherwise allows. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } export async function POST(request: Request) { diff --git a/src/app/api/oauth/codex/import/route.ts b/src/app/api/oauth/codex/import/route.ts index a7302a3a6db..6ad90072613 100644 --- a/src/app/api/oauth/codex/import/route.ts +++ b/src/app/api/oauth/codex/import/route.ts @@ -2,7 +2,7 @@ import { NextResponse } from "next/server"; import { z } from "zod"; import { normalizeCodexImportRecord, flattenCodexImportPayload } from "@/lib/oauth/services/codexImport"; import { createProviderConnection } from "@/models"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; import { refreshCodexToken, isUnrecoverableRefreshError } from "@omniroute/open-sse/services/tokenRefresh.ts"; @@ -82,10 +82,10 @@ const bodySchema = z.object({ }), }); -async function requireAuth(request: Request): Promise { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); +async function requireAuth(request: Request): Promise { + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } export async function POST(request: Request) { diff --git a/src/app/api/oauth/cursor/auto-import/route.ts b/src/app/api/oauth/cursor/auto-import/route.ts index ee0ee963f73..7a1fcdda3f4 100755 --- a/src/app/api/oauth/cursor/auto-import/route.ts +++ b/src/app/api/oauth/cursor/auto-import/route.ts @@ -1,5 +1,5 @@ import { NextResponse } from "next/server"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { tryAgentAuth, tryIdeAuth } from "@/lib/cursor/tokenExtractor"; /** @@ -11,11 +11,9 @@ import { tryAgentAuth, tryIdeAuth } from "@/lib/cursor/tokenExtractor"; * 🔒 Auth-guarded: requires JWT cookie or Bearer API key (finding #258-4). */ export async function GET(request: Request) { - if (await isAuthRequired(request)) { - if (!(await isAuthenticated(request))) { - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); - } - } + // GHSA-mg76 / GHSA-gxv4: reading/importing host credentials is a management action. + const authError = await requireManagementAuth(request, { invalidApiKeyStatus: 401 }); + if (authError) return authError; try { // Try Cursor IDE first (has both accessToken and machineId) @@ -24,6 +22,7 @@ export async function GET(request: Request) { return NextResponse.json({ found: true, accessToken: ideResult.accessToken, + refreshToken: ideResult.refreshToken, machineId: ideResult.machineId, source: ideResult.source, }); diff --git a/src/app/api/oauth/cursor/import/route.ts b/src/app/api/oauth/cursor/import/route.ts index 026cfc51cd8..89b8c267459 100755 --- a/src/app/api/oauth/cursor/import/route.ts +++ b/src/app/api/oauth/cursor/import/route.ts @@ -1,26 +1,25 @@ import { NextResponse } from "next/server"; import { CursorService } from "@/lib/oauth/services/cursor"; -import { createProviderConnection, isCloudEnabled, resolveProxyForProvider } from "@/models"; -import { getConsistentMachineId } from "@/shared/utils/machineId"; +import { credentialsFromCursorTokens } from "@/lib/oauth/services/cursorLogin"; +import { persistCursorConnection } from "@/lib/oauth/services/persistCursorConnection"; +import { isCloudEnabled } from "@/models"; import { syncToCloud } from "@/lib/cloudSync"; import { cursorImportSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; +import { getConsistentMachineId } from "@/shared/utils/machineId"; import { runWithProxyContext } from "@omniroute/open-sse/utils/proxyFetch.ts"; +import { resolveProxyForProvider } from "@/models"; async function requireOAuthImportAuth(request: Request) { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } /** * POST /api/oauth/cursor/import - * Import and validate access token from Cursor IDE's local SQLite database - * - * Request body: - * - accessToken: string - Access token from cursorAuth/accessToken - * - machineId: string - Machine ID from storage.serviceMachineId + * Import access token (and optional refresh token) from Cursor IDE / paste. */ export async function POST(request: Request) { const authResponse = await requireOAuthImportAuth(request); @@ -46,23 +45,16 @@ export async function POST(request: Request) { if (isValidationFailure(validation)) { return NextResponse.json({ error: validation.error }, { status: 400 }); } - const { accessToken, machineId } = validation.data; + const { accessToken, machineId, refreshToken } = validation.data; const cursorService = new CursorService(); - - // Resolve proxy for this provider (provider-level → global → direct) const proxy = await resolveProxyForProvider("cursor"); - // Validate token by making API call (through proxy if configured) const tokenData = await runWithProxyContext(proxy, () => cursorService.validateImportToken(accessToken.trim(), machineId?.trim()) ); - // Try to extract user info from token (JWT decode, no API call) const jwtInfo = cursorService.extractUserInfo(tokenData.accessToken); - - // Best-effort fetch real profile (email + name) from cursor.com using the - // same WorkOS session cookie format we use for usage limits. const profile = jwtInfo?.userId ? await runWithProxyContext(proxy, () => cursorService.fetchUserInfo(tokenData.accessToken, jwtInfo.userId) @@ -70,26 +62,41 @@ export async function POST(request: Request) { : null; const email = profile?.email || jwtInfo?.email || null; - - // Save to database (no `name` — let the dashboard fall back to email so the - // privacy mask toggle applies, matching the codex/claude rendering). - const connection: any = await createProviderConnection({ - provider: "cursor", - authType: "oauth", - accessToken: tokenData.accessToken, - refreshToken: null, // Cursor doesn't have public refresh endpoint - expiresAt: new Date(Date.now() + tokenData.expiresIn * 1000).toISOString(), - email, - providerSpecificData: { + const trimmedRefresh = + typeof refreshToken === "string" && refreshToken.trim().length > 0 + ? refreshToken.trim() + : null; + + let connection; + if (trimmedRefresh) { + const creds = credentialsFromCursorTokens(tokenData.accessToken, trimmedRefresh); + connection = await persistCursorConnection({ + ...creds, + email: email || creds.email, machineId: tokenData.machineId, authMethod: "imported", - provider: "Imported", - userId: jwtInfo?.userId, - }, - testStatus: "active", - }); + }); + } else { + // Access-only import — no refresh; user must re-import when expired. + const { createProviderConnection } = await import("@/models"); + connection = await createProviderConnection({ + provider: "cursor", + authType: "oauth", + accessToken: tokenData.accessToken, + refreshToken: null, + expiresAt: new Date(Date.now() + tokenData.expiresIn * 1000).toISOString(), + email, + providerSpecificData: { + machineId: tokenData.machineId, + authMethod: "imported", + provider: "Imported", + userId: jwtInfo?.userId, + accountId: jwtInfo?.userId || null, + }, + testStatus: "active", + }); + } - // Auto sync to Cloud if enabled await syncToCloudIfEnabled(); return NextResponse.json({ @@ -100,7 +107,7 @@ export async function POST(request: Request) { email: connection.email, }, }); - } catch (error: any) { + } catch (error: unknown) { console.error("Cursor import token error:", error); return NextResponse.json({ error: "Internal server error" }, { status: 500 }); } @@ -128,6 +135,12 @@ export async function GET(request: Request) { description: "From cursorAuth/accessToken in state.vscdb", type: "textarea", }, + { + name: "refreshToken", + label: "Refresh Token (optional)", + description: "From cursorAuth/refreshToken — enables automatic refresh", + type: "textarea", + }, { name: "machineId", label: "Machine ID", @@ -138,9 +151,6 @@ export async function GET(request: Request) { }); } -/** - * Sync to Cloud if enabled - */ async function syncToCloudIfEnabled() { try { const cloudEnabled = await isCloudEnabled(); diff --git a/src/app/api/oauth/cursor/login/cancel/route.ts b/src/app/api/oauth/cursor/login/cancel/route.ts new file mode 100644 index 00000000000..8f5b730f260 --- /dev/null +++ b/src/app/api/oauth/cursor/login/cancel/route.ts @@ -0,0 +1,47 @@ +import { NextResponse } from "next/server"; +import { z } from "zod"; +import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; +import { cancelCursorLoginSession } from "@/lib/oauth/services/cursorLogin"; + +const cancelSchema = z.object({ + sessionId: z.string().trim().min(1, "sessionId is required"), +}); + +async function requireOAuthAuth(request: Request) { + if (!(await isAuthRequired(request))) return null; + if (await isAuthenticated(request)) return null; + return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); +} + +/** + * POST /api/oauth/cursor/login/cancel + * Drop an in-progress deep-control login session. + */ +export async function POST(request: Request) { + const authResponse = await requireOAuthAuth(request); + if (authResponse) return authResponse; + + let rawBody: unknown; + try { + rawBody = await request.json(); + } catch { + return NextResponse.json( + { + error: { + message: "Invalid request", + details: [{ field: "body", message: "Invalid JSON body" }], + }, + }, + { status: 400 } + ); + } + + const validation = validateBody(cancelSchema, rawBody); + if (isValidationFailure(validation)) { + return NextResponse.json({ error: validation.error }, { status: 400 }); + } + + const cancelled = cancelCursorLoginSession(validation.data.sessionId); + return NextResponse.json({ success: true, cancelled }); +} diff --git a/src/app/api/oauth/cursor/login/poll/route.ts b/src/app/api/oauth/cursor/login/poll/route.ts new file mode 100644 index 00000000000..52d1bd8542c --- /dev/null +++ b/src/app/api/oauth/cursor/login/poll/route.ts @@ -0,0 +1,112 @@ +import { NextResponse } from "next/server"; +import { z } from "zod"; +import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; +import { + credentialsFromCursorTokens, + peekCursorLoginSession, + pollCursorAuthOnce, + consumeCursorLoginSession, +} from "@/lib/oauth/services/cursorLogin"; +import { persistCursorConnection } from "@/lib/oauth/services/persistCursorConnection"; +import { isCloudEnabled } from "@/models"; +import { syncToCloud } from "@/lib/cloudSync"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; +import { getConsistentMachineId } from "@/shared/utils/machineId"; + +const pollSchema = z.object({ + sessionId: z.string().trim().min(1, "sessionId is required"), +}); + +async function requireOAuthAuth(request: Request) { + if (!(await isAuthRequired(request))) return null; + if (await isAuthenticated(request)) return null; + return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); +} + +async function syncToCloudIfEnabled() { + try { + if (await isCloudEnabled()) { + await syncToCloud(); + } + } catch { + // best-effort + } +} + +/** + * POST /api/oauth/cursor/login/poll + * One poll against Cursor auth/poll. UI repeats until ok/error/timeout. + */ +export async function POST(request: Request) { + const authResponse = await requireOAuthAuth(request); + if (authResponse) return authResponse; + + let rawBody: unknown; + try { + rawBody = await request.json(); + } catch { + return NextResponse.json( + { + error: { + message: "Invalid request", + details: [{ field: "body", message: "Invalid JSON body" }], + }, + }, + { status: 400 } + ); + } + + const validation = validateBody(pollSchema, rawBody); + if (isValidationFailure(validation)) { + return NextResponse.json({ error: validation.error }, { status: 400 }); + } + + const { sessionId } = validation.data; + const session = peekCursorLoginSession(sessionId); + if (!session) { + return NextResponse.json( + { status: "expired", error: "Login session expired or not found. Start again." }, + { status: 410 } + ); + } + + try { + const result = await pollCursorAuthOnce(session.uuid, session.verifier); + if (result.status === "pending") { + return NextResponse.json({ status: "pending" }); + } + if (result.status === "error") { + return NextResponse.json( + { status: "error", error: result.message }, + { status: result.httpStatus && result.httpStatus >= 400 ? result.httpStatus : 502 } + ); + } + + // Success — consume session so verifier cannot be reused + consumeCursorLoginSession(sessionId); + + const creds = credentialsFromCursorTokens(result.accessToken, result.refreshToken); + const machineId = await getConsistentMachineId(); + const connection = await persistCursorConnection({ + ...creds, + machineId, + authMethod: "deep_control", + }); + + await syncToCloudIfEnabled(); + + return NextResponse.json({ + status: "ok", + success: true, + connection: { + id: (connection as { id?: string })?.id, + provider: "cursor", + email: (connection as { email?: string })?.email ?? creds.email ?? null, + }, + }); + } catch (error) { + const message = sanitizeErrorMessage(error) || "Failed to poll Cursor login"; + return NextResponse.json({ error: message }, { status: 500 }); + } +} diff --git a/src/app/api/oauth/cursor/login/start/route.ts b/src/app/api/oauth/cursor/login/start/route.ts new file mode 100644 index 00000000000..24041ec49ec --- /dev/null +++ b/src/app/api/oauth/cursor/login/start/route.ts @@ -0,0 +1,37 @@ +import { NextResponse } from "next/server"; +import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { + createCursorLoginSession, + generateCursorAuthParams, +} from "@/lib/oauth/services/cursorLogin"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; + +async function requireOAuthAuth(request: Request) { + if (!(await isAuthRequired(request))) return null; + if (await isAuthenticated(request)) return null; + return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); +} + +/** + * POST /api/oauth/cursor/login/start + * Begin deep-control PKCE login. Verifier stays server-side. + */ +export async function POST(request: Request) { + const authResponse = await requireOAuthAuth(request); + if (authResponse) return authResponse; + + try { + const params = await generateCursorAuthParams(); + const { sessionId, loginUrl } = createCursorLoginSession(params); + return NextResponse.json({ + success: true, + sessionId, + loginUrl, + // Multi-replica note: sessions are in-process; use sticky routing if scaled out. + expiresInSeconds: 15 * 60, + }); + } catch (error) { + const message = sanitizeErrorMessage(error) || "Failed to start Cursor login"; + return NextResponse.json({ error: message }, { status: 500 }); + } +} diff --git a/src/app/api/oauth/kiro/auto-import/route.ts b/src/app/api/oauth/kiro/auto-import/route.ts index 40b927b4c9d..ca61b177ddb 100755 --- a/src/app/api/oauth/kiro/auto-import/route.ts +++ b/src/app/api/oauth/kiro/auto-import/route.ts @@ -1,7 +1,7 @@ import { NextResponse } from "next/server"; import { homedir } from "os"; import { join } from "path"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { createProviderConnection, getProviderConnections, @@ -31,11 +31,9 @@ import { * 🔒 Auth-guarded: requires JWT cookie or Bearer API key. */ export async function GET(request: Request) { - if (await isAuthRequired(request)) { - if (!(await isAuthenticated(request))) { - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); - } - } + // GHSA-mg76 / GHSA-gxv4: reading/importing host credentials is a management action. + const authError = await requireManagementAuth(request, { invalidApiKeyStatus: 401 }); + if (authError) return authError; const { searchParams } = new URL(request.url); const targetProvider = searchParams.get("targetProvider") === "amazon-q" ? "amazon-q" : "kiro"; diff --git a/src/app/api/oauth/kiro/import/route.ts b/src/app/api/oauth/kiro/import/route.ts index d4b89183abd..29aa19d07fc 100755 --- a/src/app/api/oauth/kiro/import/route.ts +++ b/src/app/api/oauth/kiro/import/route.ts @@ -11,7 +11,7 @@ import { getConsistentMachineId } from "@/shared/utils/machineId"; import { syncToCloud } from "@/lib/cloudSync"; import { kiroImportSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { runWithProxyContext } from "@omniroute/open-sse/utils/proxyFetch.ts"; import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error"; import { findKiroConnectionByIdentity } from "@/lib/oauth/kiroConnectionIdentity"; @@ -38,9 +38,9 @@ export function buildKiroImportError(error: unknown): string { } async function requireOAuthImportAuth(request: Request) { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } async function upsertImportedKiroConnection( diff --git a/src/app/api/oauth/raycast/auto-import/route.ts b/src/app/api/oauth/raycast/auto-import/route.ts index 17e4cf48c51..4dc3ac80018 100644 --- a/src/app/api/oauth/raycast/auto-import/route.ts +++ b/src/app/api/oauth/raycast/auto-import/route.ts @@ -14,14 +14,14 @@ import { extractLocalRaycastCredentials, isRaycastLocalExtractAvailable, } from "@/lib/oauth/services/raycastLocal"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { resolveProxyForProvider } from "@/models"; import { runWithProxyContext } from "@omniroute/open-sse/utils/proxyFetch.ts"; async function requireOAuthImportAuth(request: Request) { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } export async function GET(request: Request) { diff --git a/src/app/api/oauth/raycast/import/route.ts b/src/app/api/oauth/raycast/import/route.ts index 0ff02764fab..266dd1aecc6 100644 --- a/src/app/api/oauth/raycast/import/route.ts +++ b/src/app/api/oauth/raycast/import/route.ts @@ -11,14 +11,14 @@ import { createProviderConnection } from "@/models"; import { RaycastService } from "@/lib/oauth/services/raycast"; import { raycastImportSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { resolveProxyForProvider } from "@/models"; import { runWithProxyContext } from "@omniroute/open-sse/utils/proxyFetch.ts"; async function requireOAuthImportAuth(request: Request) { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } export async function POST(request: Request) { diff --git a/src/app/api/oauth/trae/import/route.ts b/src/app/api/oauth/trae/import/route.ts index 9f3fba5bacb..c3faeb8e901 100644 --- a/src/app/api/oauth/trae/import/route.ts +++ b/src/app/api/oauth/trae/import/route.ts @@ -2,7 +2,7 @@ import { NextResponse } from "next/server"; import { createProviderConnection } from "@/models"; import { traeImportSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; -import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; /** * POST /api/oauth/trae/import @@ -22,9 +22,9 @@ import { isAuthRequired, isAuthenticated } from "@/shared/utils/apiAuth"; * region — optional, default "US-East" */ async function requireOAuthImportAuth(request: Request) { - if (!(await isAuthRequired(request))) return null; - if (await isAuthenticated(request)) return null; - return NextResponse.json({ error: "Unauthorized" }, { status: 401 }); + // GHSA-mg76: importing a provider connection is a state-mutating admin action; + // require management scope (or a dashboard session), not any valid client key. + return requireManagementAuth(request, { invalidApiKeyStatus: 401 }); } export async function POST(request: Request) { diff --git a/src/app/api/provider-models/route.ts b/src/app/api/provider-models/route.ts index 6b286b5b19d..d6b672aae61 100644 --- a/src/app/api/provider-models/route.ts +++ b/src/app/api/provider-models/route.ts @@ -28,6 +28,7 @@ import { isAnthropicCompatibleProvider, } from "@/shared/constants/providers"; import { isAuthenticated } from "@/shared/utils/apiAuth"; +export const dynamic = "force-dynamic"; import { providerModelMutationSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; diff --git a/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts b/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts index 62988798d77..0939b59b554 100644 --- a/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts +++ b/src/app/api/providers/[id]/models/discovery/providerModelsConfig.ts @@ -87,6 +87,22 @@ export function parseAlibabaModelStudioModelsForConnection( export function parseQwenCloudTextModels(data: any): any[] { return parseCuratedDashscopeModels(data, QWEN_CLOUD_TEXT_MODELS, QWEN_CLOUD_TEXT_MODEL_IDS); } + +// Perplexity's /v1/models lists the Agent API catalog (vendor-prefixed ids like +// "anthropic/claude-fable-5"), but chat requests always go to the classic +// /chat/completions endpoint, which only accepts the Sonar family. Filter +// discovery to Sonar-family ids so agent-style ids never surface as routable +// chat models (#11060). Bounded pattern — no ReDoS-prone quantifiers. +export function parsePerplexitySonarModels(data: any): any[] { + const models = Array.isArray(data?.data) + ? data.data + : Array.isArray(data?.models) + ? data.models + : []; + return models.filter( + (model: any) => typeof model?.id === "string" && /^sonar(-|$)/.test(model.id) + ); +} type ProviderModelsHeaderContext = { authType?: string; providerSpecificData?: unknown; @@ -659,6 +675,17 @@ export const PROVIDER_MODELS_CONFIG: Record = headers: { Accept: "application/json" }, parseResponse: parseClinepassRecommendedModels, }, + // Perplexity's /v1/models lists the Agent API catalog (vendor-prefixed agent + // ids), but chat only accepts the Sonar family on /chat/completions. Import + // must keep Sonar-family ids only (#11060). + perplexity: { + url: "https://api.perplexity.ai/v1/models", + method: "GET", + headers: { "Content-Type": "application/json" }, + authHeader: "Authorization", + authPrefix: "Bearer ", + parseResponse: parsePerplexitySonarModels, + }, cohere: { url: "https://api.cohere.com/v2/models", method: "GET", diff --git a/src/app/api/providers/[id]/models/route.ts b/src/app/api/providers/[id]/models/route.ts index 818156aaec1..01f1bd72857 100755 --- a/src/app/api/providers/[id]/models/route.ts +++ b/src/app/api/providers/[id]/models/route.ts @@ -92,6 +92,7 @@ import { import { getSyncedAvailableModels, getCustomModels } from "@/lib/db/models"; import { isConnectionUnavailableToAuxiliaryActivity } from "@/lib/exclusiveLeaseIsolation"; import { fetchCursorAgentModels } from "@/lib/providerModels/cursorAgent"; +import { fetchCursorAvailableModels } from "@/lib/providerModels/cursorAvailableModels"; import { ensureCursorAutoCatalogEntry } from "@/lib/providerModels/cursorAutoCatalog"; import { fetchRaycastModels } from "@omniroute/open-sse/services/raycast.ts"; import { runWithProxyContext } from "@omniroute/open-sse/utils/proxyFetch.ts"; @@ -1356,19 +1357,45 @@ export async function GET( const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled(); if (autoFetchDisabledResponse) return autoFetchDisabledResponse; + const warnings: string[] = []; + const token = (accessToken || apiKey || "").trim(); + const machineId = + typeof connection?.providerSpecificData === "object" && + connection.providerSpecificData && + typeof (connection.providerSpecificData as { machineId?: unknown }).machineId === "string" + ? (connection.providerSpecificData as { machineId: string }).machineId + : null; + + if (token) { + try { + const models = await fetchCursorAvailableModels({ + accessToken: token, + machineId, + }); + return buildApiDiscoveryResponse(models); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + console.log("[models] Cursor AvailableModels failed:", message); + warnings.push(`AvailableModels unavailable (${message})`); + } + } else { + warnings.push("no Cursor access token on connection"); + } + try { const models = ensureCursorAutoCatalogEntry(await fetchCursorAgentModels()); return buildApiDiscoveryResponse(models); } catch (err) { const message = err instanceof Error ? err.message : String(err); console.log("[models] cursor-agent fetch failed:", message); + const detail = [...warnings, `cursor-agent unavailable (${message})`].join("; "); const fallback = buildDiscoveryFallbackResponse({ - cacheWarning: `cursor-agent unavailable (${message}) — using cached catalog`, - localWarning: `cursor-agent unavailable (${message}) — using local catalog`, + cacheWarning: `${detail} — using cached catalog`, + localWarning: `${detail} — using local catalog`, }); if (fallback) return fallback; return NextResponse.json( - { error: `Failed to fetch Cursor models: ${message}` }, + { error: `Failed to fetch Cursor models: ${detail}` }, { status: 502 } ); } diff --git a/src/app/api/providers/[id]/route.ts b/src/app/api/providers/[id]/route.ts index 562dad07444..4f588571b87 100644 --- a/src/app/api/providers/[id]/route.ts +++ b/src/app/api/providers/[id]/route.ts @@ -121,7 +121,17 @@ export async function PUT(request: Request, { params }: { params: Promise<{ id: const { id } = await params; const validation = validateBody(updateProviderConnectionSchema, rawBody); if (isValidationFailure(validation)) { - return NextResponse.json({ error: validation.error }, { status: 400 }); + // never drop an operator's intent silently. Surface the rejected + // keys (field paths and unrecognized-key names) alongside the existing + // error envelope so clients and the UI can tell exactly what was refused. + const rejected = [ + ...validation.error.details.map((d) => d.field).filter(Boolean), + ...validation.error.details.flatMap((d) => d.keys ?? []), + ]; + return NextResponse.json( + { error: { ...validation.error, rejected } }, + { status: 400 } + ); } const body = validation.data; const { diff --git a/src/app/api/providers/route.ts b/src/app/api/providers/route.ts index f1825674c25..9ed7f10667f 100644 --- a/src/app/api/providers/route.ts +++ b/src/app/api/providers/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { getAuditRequestContext, logAuditEvent } from "@/lib/compliance/index"; import { getProviderAuditTarget, @@ -138,8 +139,6 @@ export async function POST(request: Request) { let providerSpecificData = incomingPsd || null; let persistedApiKey = apiKey; - const allowMultipleCompatibleConnections = - process.env.ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE === "true"; if (provider === "qoder") { providerSpecificData = normalizeQoderPatProviderData(providerSpecificData || {}); @@ -174,7 +173,6 @@ export async function POST(request: Request) { return NextResponse.json({ error: "OpenAI Compatible node not found" }, { status: 404 }); } - const existingConnections = await getProviderConnections({ provider }); // Allow multiple connections for compatible nodes exactly like first-party providers providerSpecificData = { @@ -200,7 +198,6 @@ export async function POST(request: Request) { ); } - const existingConnections = await getProviderConnections({ provider }); // Allow multiple connections for compatible nodes exactly like first-party providers providerSpecificData = { diff --git a/src/app/api/search/providers/route.ts b/src/app/api/search/providers/route.ts index d838e749a57..31bada96faf 100644 --- a/src/app/api/search/providers/route.ts +++ b/src/app/api/search/providers/route.ts @@ -1,7 +1,7 @@ import { NextResponse } from "next/server"; import { SEARCH_PROVIDERS, - SEARCH_CREDENTIAL_FALLBACKS, + getSearchCredentialFallbacks, } from "@omniroute/open-sse/config/searchRegistry.ts"; import { isAuthenticated } from "@/shared/utils/apiAuth"; import { getProviderCredentials } from "@/sse/services/auth"; @@ -85,8 +85,7 @@ async function resolveProviderStatus( // All rate limited — check fallback before returning rate_limited if (isAllRateLimitedCredentials(credentials)) { if (useCredentialFallback) { - const fallbackId = SEARCH_CREDENTIAL_FALLBACKS[providerId]; - if (fallbackId) { + for (const fallbackId of getSearchCredentialFallbacks(providerId)) { const fallbackCreds = await getProviderCredentials(fallbackId).catch(() => null); if (fallbackCreds && !isAllRateLimitedCredentials(fallbackCreds)) { return "configured"; @@ -98,16 +97,17 @@ async function resolveProviderStatus( // null → no credentials; try fallback if (useCredentialFallback) { - const fallbackId = SEARCH_CREDENTIAL_FALLBACKS[providerId]; - if (fallbackId) { + let fallbackRateLimited = false; + for (const fallbackId of getSearchCredentialFallbacks(providerId)) { const fallbackCreds = await getProviderCredentials(fallbackId).catch(() => null); if (fallbackCreds && !isAllRateLimitedCredentials(fallbackCreds)) { return "configured"; } if (isAllRateLimitedCredentials(fallbackCreds)) { - return "rate_limited"; + fallbackRateLimited = true; } } + if (fallbackRateLimited) return "rate_limited"; } return "missing"; diff --git a/src/app/api/settings/obsidian/webdav/route.ts b/src/app/api/settings/obsidian/webdav/route.ts index 1465b1dafd7..4b098cfdb00 100644 --- a/src/app/api/settings/obsidian/webdav/route.ts +++ b/src/app/api/settings/obsidian/webdav/route.ts @@ -1,6 +1,7 @@ import { NextRequest, NextResponse } from "next/server"; import { z } from "zod"; import { isAuthenticated } from "@/shared/utils/apiAuth"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { buildErrorBody } from "@omniroute/open-sse/utils/error"; import { getObsidianSyncStatus, @@ -21,10 +22,19 @@ export async function GET(request: NextRequest) { try { const status = await getObsidianSyncStatus(); + // GHSA-62vw: the WebDAV password is reusable authentication material. Return + // the plaintext only to a genuine management principal (dashboard session or + // manage-scope key), never to an anonymous caller that reached this handler + // through the requireLogin=false open mode. The dashboard's authenticated + // reveal-password view is unaffected; anonymous callers get a set/unset flag. + const hasManagement = + (await requireManagementAuth(request, { alwaysRequireAuth: true })) === null; return NextResponse.json({ webdavEnabled: status.webdavEnabled, webdavUsername: status.webdavEnabled ? status.webdavUsername : null, - webdavPassword: status.webdavEnabled ? status.webdavPassword : null, + webdavPassword: + status.webdavEnabled && hasManagement ? status.webdavPassword : null, + webdavPasswordSet: status.webdavEnabled && Boolean(status.webdavPassword), vaultPath: status.vaultPath, }); } catch (error) { diff --git a/src/app/api/usage/analytics/route.ts b/src/app/api/usage/analytics/route.ts index 306292865a2..533733c767c 100644 --- a/src/app/api/usage/analytics/route.ts +++ b/src/app/api/usage/analytics/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { getProviderById } from "@/shared/constants/providers"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { getApiKeys } from "@/lib/db/apiKeys"; diff --git a/src/app/api/usage/call-logs/route.ts b/src/app/api/usage/call-logs/route.ts index 0a737ea1f96..c91a0561a5b 100644 --- a/src/app/api/usage/call-logs/route.ts +++ b/src/app/api/usage/call-logs/route.ts @@ -1,8 +1,9 @@ import { NextResponse } from "next/server"; +export const dynamic = "force-dynamic"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { getCallLogs } from "@/lib/usageDb"; import { getCompletedDetails, getPendingById } from "@/lib/usage/usageHistory"; -import { getProviderConnections } from "@/lib/localDb"; +import { getProviderConnections } from "@/lib/db/providers"; import { getProviderNodes } from "@/models"; import { matchesSearch } from "@/shared/utils/turkishText"; @@ -26,6 +27,66 @@ function rowPriority(row: any): number { return 2; } +/** + * Applies the active filter predicates to a single merged call-log row. + * + * `getCallLogs()` already filters the persisted DB rows server-side, but the + * in-memory entries (active/pending + recently-completed) are merged in by + * `buildCallLogListRows()` and would otherwise bypass every filter except + * `correlationId`. Running the same predicates over the merged rows closes that + * gap. It is idempotent for DB rows (they already satisfy the predicate) while + * correctly excluding in-memory rows that do not match. + */ +export function rowMatchesFilter(row: any, filter: Record): boolean { + if (!filter) return true; + + if (filter.status === "error") { + if (!(Number(row?.status) >= 400 || Boolean(row?.error))) return false; + } else if (filter.status === "ok") { + if (!(Number(row?.status) >= 200 && Number(row?.status) < 300)) return false; + } else if (typeof filter.status === "number" || (typeof filter.status === "string" && !isNaN(Number(filter.status)))) { + if (Number(row?.status) !== Number(filter.status)) return false; + } + + if (filter.model && !matchesSearch(row?.model || "", String(filter.model))) { + return false; + } + if (filter.provider && !matchesSearch(row?.provider || "", String(filter.provider))) { + return false; + } + if (filter.account && !matchesSearch(row?.account || "", String(filter.account))) { + return false; + } + if (filter.apiKey && !matchesSearch(row?.apiKeyName || "", String(filter.apiKey))) { + return false; + } + if (filter.combo && !matchesSearch(row?.comboName || "", String(filter.combo))) { + return false; + } + if (filter.correlationId && !matchesSearch(row?.correlationId || "", String(filter.correlationId))) { + return false; + } + if (filter.search) { + const term = String(filter.search); + const haystack = [ + row?.model, + row?.provider, + row?.providerDisplay, + row?.account, + row?.apiKeyName, + row?.comboName, + row?.correlationId, + row?.error, + row?.path, + ] + .filter(Boolean) + .join(" "); + if (!matchesSearch(haystack, term)) return false; + } + + return true; +} + export function buildCallLogListRows({ logs, connections, @@ -173,15 +234,8 @@ export async function GET(request: Request) { completedDetails: getCompletedDetails().values(), }); - // When correlationId filter is set, also filter in-memory entries - // (active + completed) that don't match — getCallLogs already filters - // the DB rows but activeEntries/completedEntries bypass it. - if (filter.correlationId) { - const cid = filter.correlationId; - return NextResponse.json(rows.filter((r: any) => matchesSearch(r.correlationId || "", cid))); - } - - return NextResponse.json(rows); + const filtered = rows.filter((r: any) => rowMatchesFilter(r, filter)); + return NextResponse.json(filtered); } catch (error) { console.error("[API ERROR] /api/usage/call-logs failed:", error); return NextResponse.json({ error: "Failed to fetch call logs" }, { status: 500 }); diff --git a/src/app/api/usage/utilization/route.ts b/src/app/api/usage/utilization/route.ts index 40b84541425..23ce913ac33 100644 --- a/src/app/api/usage/utilization/route.ts +++ b/src/app/api/usage/utilization/route.ts @@ -1,6 +1,6 @@ import { NextResponse } from "next/server"; import { getAggregatedSnapshots } from "@/lib/db/quotaSnapshots"; -import { getConnection } from "@/lib/db/connections"; +import { getProviderConnectionById } from "@/lib/db/providers"; import type { ProviderUtilizationResponse, UtilizationTimeRange, @@ -10,6 +10,10 @@ import { BUCKET_SIZES } from "@/shared/types/utilization"; const VALID_RANGES: UtilizationTimeRange[] = ["1h", "24h", "7d", "30d"]; +function asNullableString(value: unknown): string | null { + return typeof value === "string" ? value : null; +} + function getRangeStartIso(range: UtilizationTimeRange): string { const end = new Date(); const start = new Date(end); @@ -67,11 +71,11 @@ export async function GET(request: Request) { ); connectionMeta = {}; for (const cid of uniqueConnectionIds) { - const conn = getConnection(cid); + const conn = await getProviderConnectionById(cid); connectionMeta[cid] = { - email: conn?.email ?? null, - name: conn?.name ?? null, - displayName: conn?.displayName ?? null, + email: asNullableString(conn?.email), + name: asNullableString(conn?.name), + displayName: asNullableString(conn?.displayName), }; } } diff --git a/src/app/api/v1/combos/projectCombo.ts b/src/app/api/v1/combos/projectCombo.ts index 0fdae64ef89..924022d505f 100644 --- a/src/app/api/v1/combos/projectCombo.ts +++ b/src/app/api/v1/combos/projectCombo.ts @@ -5,6 +5,10 @@ * returning combo metadata to API-key callers. Kept in a separate module so * the projection can be unit-tested without spinning up the Next.js route. * + * #10968: the projection also reports `accountPinned` per model step — a boolean + * derived from the stripped `connectionId`, so callers can distinguish a combo + * that fails over between two accounts of one provider from a duplicated step. + * * #3979: client-facing combo catalogs (the `/v1/combos`, VS Code and LobeHub / * OpenCode import surfaces) can opt into advertising the combo's resolved * capabilities (multimodal / reasoning / caching) so importing clients enable @@ -17,6 +21,19 @@ export interface PublicComboStep { model?: string; comboName?: string; providerId?: string; + /** + * #10968: whether this step pins one specific account of its provider. + * + * Two steps that pin different accounts of the same provider project to + * identical `{kind, model, providerId}` objects, so a client cannot tell a + * two-account failover from the same step listed twice. This says which it + * is without exposing the `connectionId` the flag is derived from — not even + * a prefix, per the issue. + * + * Set on every `model` step. Absent on `combo-ref`, which routes through + * another combo and has no account of its own. + */ + accountPinned?: boolean; } /** @@ -66,6 +83,10 @@ export function projectComboStep(step: Record): PublicComboStep if (typeof step.providerId === "string" && step.providerId.length > 0) { out.providerId = step.providerId; } + // Same shape test as providerId above. `cleanupComboConnectionRefs` drops the + // key when the connection is deleted, so a step whose pinned account is gone + // reports false rather than pointing at nothing. + out.accountPinned = typeof step.connectionId === "string" && step.connectionId.length > 0; return out; } return null; diff --git a/src/app/api/v1/moderations/route.ts b/src/app/api/v1/moderations/route.ts index 36fb4aa75a8..1ab740e5adc 100644 --- a/src/app/api/v1/moderations/route.ts +++ b/src/app/api/v1/moderations/route.ts @@ -68,7 +68,7 @@ async function postHandler(request, context) { const response = await handleModeration({ body: { ...body, model }, credentials }); if (response?.ok) { - await clearRecoveredProviderState(credentials); + await clearRecoveredProviderState(credentials as Record); } return response; } diff --git a/src/app/api/v1/rerank/route.ts b/src/app/api/v1/rerank/route.ts index bf9da386eb4..04fb1632dc6 100644 --- a/src/app/api/v1/rerank/route.ts +++ b/src/app/api/v1/rerank/route.ts @@ -10,11 +10,15 @@ import { HTTP_STATUS } from "@omniroute/open-sse/config/constants.ts"; import { enforceApiKeyPolicy } from "@/shared/utils/apiKeyPolicy"; import { v1RerankSchema } from "@/shared/validation/schemas"; import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; -import { getCachedProviderNodes } from "@/lib/localDb"; +import { getCachedProviderNodes } from "@/lib/db/readCache"; import { isAllRateLimitedCredentials, rateLimitedProviderResponse, } from "@/app/api/v1/_shared/rateLimit"; +import { saveCallLog } from "@/lib/usageDb"; +import { attachOmniRouteMetaHeaders } from "@/domain/omnirouteResponseMeta"; +import { generateRequestId } from "@/shared/utils/requestId"; +import { CORS_HEADERS } from "@omniroute/open-sse/utils/cors.ts"; /** * Handle CORS preflight @@ -121,6 +125,8 @@ async function postHandler(request, context) { return_documents: body.return_documents, credentials, connectionId: (credentials as { connectionId?: string } | null)?.connectionId || null, + apiKeyId: policy.apiKeyInfo?.id || null, + apiKeyName: policy.apiKeyInfo?.name || null, }); if (response?.ok) { await clearRecoveredProviderState(credentials); @@ -148,8 +154,9 @@ async function postHandler(request, context) { } const token = credentials?.apiKey || credentials?.accessToken; + const startTime = Date.now(); try { - const res = await fetch(localProvider.baseUrl, { + let res = await fetch(localProvider.baseUrl, { method: "POST", headers: { "Content-Type": "application/json", @@ -164,19 +171,110 @@ async function postHandler(request, context) { }), }); + // Some local providers (e.g. Infinity, TEI) mount at /rerank rather than /v1/rerank + if (res.status === 404 && localProvider.baseUrl.endsWith("/v1/rerank")) { + const fallbackUrl = localProvider.baseUrl.replace(/\/v1\/rerank$/, "/rerank"); + try { + const fallbackRes = await fetch(fallbackUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }, + body: JSON.stringify({ + model: localModel, + query: body.query, + documents: body.documents, + top_n: body.top_n || body.documents.length, + return_documents: body.return_documents !== false, + }), + }); + if (fallbackRes.ok || fallbackRes.status !== 404) { + res = fallbackRes; + } + } catch { + // retain original 404 response if fallback fetch fails + } + } + if (!res.ok) { const errData = await res.json().catch(() => ({})); - return errorResponse( - res.status, - errData.message || errData.detail || `Provider returned HTTP ${res.status}` - ); + const errorMessage = + errData.message || errData.detail || `Provider returned HTTP ${res.status}`; + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: res.status, + model: body.model, + provider: prefix, + connectionId: + (credentials as { connectionId?: string } | null)?.connectionId || undefined, + duration: Date.now() - startTime, + requestBody: { + model: body.model, + query: body.query, + documents: body.documents, + top_n: body.top_n, + return_documents: body.return_documents, + }, + responseBody: errData, + error: errorMessage, + apiKeyId: policy.apiKeyInfo?.id || undefined, + apiKeyName: policy.apiKeyInfo?.name || undefined, + }).catch(() => {}); + return errorResponse(res.status, errorMessage); } const data = await res.json(); - return Response.json(data, { - headers: {}, + const latencyMs = Date.now() - startTime; + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: 200, + model: body.model, + provider: prefix, + connectionId: + (credentials as { connectionId?: string } | null)?.connectionId || undefined, + duration: latencyMs, + tokens: { prompt_tokens: 0, completion_tokens: 0 }, + requestBody: { + model: body.model, + query: body.query, + documents: body.documents, + top_n: body.top_n, + return_documents: body.return_documents, + }, + responseBody: data, + apiKeyId: policy.apiKeyInfo?.id || undefined, + apiKeyName: policy.apiKeyInfo?.name || undefined, + }).catch(() => {}); + + const headers = new Headers({ ...CORS_HEADERS, "Content-Type": "application/json" }); + attachOmniRouteMetaHeaders(headers, { + provider: prefix, + model: localModel, + costUsd: 0, + latencyMs, + requestId: generateRequestId(), + }); + return new Response(JSON.stringify(data), { + status: 200, + headers, }); } catch (err: any) { + saveCallLog({ + method: "POST", + path: "/v1/rerank", + status: 500, + model: body.model, + provider: prefix, + connectionId: + (credentials as { connectionId?: string } | null)?.connectionId || undefined, + duration: Date.now() - startTime, + error: err.message, + apiKeyId: policy.apiKeyInfo?.id || undefined, + apiKeyName: policy.apiKeyInfo?.name || undefined, + }).catch(() => {}); return errorResponse(500, `Rerank request failed: ${err.message}`); } } diff --git a/src/app/api/v1/search/route.ts b/src/app/api/v1/search/route.ts index 0158e4947c9..313b72dece0 100644 --- a/src/app/api/v1/search/route.ts +++ b/src/app/api/v1/search/route.ts @@ -10,8 +10,9 @@ import { resolveSearchProvider, selectProvider, supportsSearchType, + isUnconfiguredLoopbackSearchProvider, SEARCH_PROVIDERS, - SEARCH_CREDENTIAL_FALLBACKS, + getSearchCredentialFallbacks, } from "@omniroute/open-sse/config/searchRegistry.ts"; import { errorResponse } from "@omniroute/open-sse/utils/error.ts"; import { HTTP_STATUS } from "@omniroute/open-sse/config/constants.ts"; @@ -57,9 +58,7 @@ export async function OPTIONS() { export async function GET() { const settings = await getSettings().catch(() => ({} as any)); const blockedProviders = settings?.blockedProviders || []; - const providers = getAllSearchProviders().filter( - (p) => !isProviderBlockedByIdOrAlias(p.id, blockedProviders) - ); + const providers = getAllSearchProviders(blockedProviders); const timestamp = Math.floor(Date.now() / 1000); const data = providers.map((p) => ({ @@ -82,17 +81,17 @@ async function resolveSearchCredentials(providerId: string): Promise null); if (credentials && !isAllRateLimitedCredentials(credentials)) return credentials; - const fallbackId = SEARCH_CREDENTIAL_FALLBACKS[providerId]; - if (!fallbackId) return credentials; - - const fallbackCredentials = await getProviderCredentialsWithQuotaPreflight(fallbackId).catch( - () => null - ); - if (fallbackCredentials && !isAllRateLimitedCredentials(fallbackCredentials)) { - return fallbackCredentials; + for (const fallbackId of getSearchCredentialFallbacks(providerId)) { + const fallbackCredentials = await getProviderCredentialsWithQuotaPreflight(fallbackId).catch( + () => null + ); + if (fallbackCredentials && !isAllRateLimitedCredentials(fallbackCredentials)) { + return fallbackCredentials; + } + if (fallbackCredentials) return fallbackCredentials; } - return fallbackCredentials || credentials; + return credentials; } async function resolveSearchExecutionCredentials(providerConfig: { @@ -133,6 +132,8 @@ async function postHandler(request: Request, context: unknown) { return errorResponse(HTTP_STATUS.BAD_REQUEST, formatValidationMessage(validation.error)); } const body = validation.data; + if (body.provider === "x_search") body.provider = "x-search"; + if (body.provider === "x-search") body.search_type = "x"; // Enforce API key policies — use "search" as model identifier for consistent policy config const policy = await enforceApiKeyPolicy(request, "search"); @@ -286,6 +287,7 @@ async function postHandler(request: Request, context: unknown) { if (!alternateProviderId) { for (const provider of Object.values(SEARCH_PROVIDERS)) { if (!provider.fallbackOnly || provider.id === providerConfig.id) continue; + if (isUnconfiguredLoopbackSearchProvider(provider)) continue; if (!supportsSearchType(provider, body.search_type)) continue; const fallbackCreds = await resolveSearchExecutionCredentials(provider); if (fallbackCreds && !isAllRateLimitedCredentials(fallbackCreds)) { diff --git a/src/app/api/webhooks/[id]/route.ts b/src/app/api/webhooks/[id]/route.ts index 5490da1ed41..98fc6e79c75 100644 --- a/src/app/api/webhooks/[id]/route.ts +++ b/src/app/api/webhooks/[id]/route.ts @@ -15,22 +15,15 @@ import { encryptMetadata } from "@/lib/webhookDispatcher"; import { isEncryptionEnabled } from "@/lib/db/encryption"; import { parseAndValidateWebhookUrl } from "@/shared/network/outboundUrlGuardPolicy"; +import { WEBHOOK_EVENT_VALUES } from "@/lib/webhooks/eventDescriptions"; + const WEBHOOK_KINDS = ["slack", "telegram", "discord", "custom"] as const; -const WEBHOOK_EVENT_VALUES = [ - "*", - "request.completed", - "request.failed", - "provider.error", - "provider.recovered", - "quota.exceeded", - "combo.switched", - "test.ping", -] as const; +const WEBHOOK_EVENT_VALUES_WITH_WILDCARD = ["*", ...WEBHOOK_EVENT_VALUES] as const; const updateWebhookSchema = z .object({ url: z.string().min(1).max(2000).optional(), - events: z.array(z.enum(WEBHOOK_EVENT_VALUES)).optional(), + events: z.array(z.enum(WEBHOOK_EVENT_VALUES_WITH_WILDCARD)).optional(), secret: z.string().max(500).optional(), description: z.string().max(1000).optional(), enabled: z.boolean().optional(), diff --git a/src/app/api/webhooks/route.ts b/src/app/api/webhooks/route.ts index 44e3d7b6348..2cb018e13fb 100644 --- a/src/app/api/webhooks/route.ts +++ b/src/app/api/webhooks/route.ts @@ -14,12 +14,15 @@ import { encryptMetadata } from "@/lib/webhookDispatcher"; import { isEncryptionEnabled } from "@/lib/db/encryption"; import { parseAndValidateWebhookUrl } from "@/shared/network/outboundUrlGuardPolicy"; +import { WEBHOOK_EVENT_VALUES } from "@/lib/webhooks/eventDescriptions"; + const WEBHOOK_KINDS = ["slack", "telegram", "discord", "custom"] as const; +const WEBHOOK_EVENT_VALUES_WITH_WILDCARD = ["*", ...WEBHOOK_EVENT_VALUES] as const; const createWebhookSchema = z .object({ url: z.string().min(1).max(2000), - events: z.array(z.string()).optional().default(["*"]), + events: z.array(z.enum(WEBHOOK_EVENT_VALUES_WITH_WILDCARD)).optional().default(["*"]), secret: z.string().max(500).optional(), description: z.string().max(1000).optional().default(""), kind: z.enum(WEBHOOK_KINDS).optional().default("custom"), diff --git a/src/app/readyz/route.ts b/src/app/readyz/route.ts new file mode 100644 index 00000000000..796db3b2158 --- /dev/null +++ b/src/app/readyz/route.ts @@ -0,0 +1,6 @@ +/** + * Kubernetes-style readiness alias of /healthz. + * Same lifecycle phase, same 200/503 bodies. Not a liveness probe. + */ +export const dynamic = "force-dynamic"; +export { GET, HEAD } from "../healthz/route"; diff --git a/src/domain/configAudit.ts b/src/domain/configAudit.ts index c2701b2a21f..873d3b046a5 100644 --- a/src/domain/configAudit.ts +++ b/src/domain/configAudit.ts @@ -13,6 +13,8 @@ * - Optional human notes */ +import { getDbInstance } from "../lib/db/core"; + /** Types of configuration entities that can be audited */ export type AuditTarget = "provider" | "combo" | "policy" | "connection" | "settings"; @@ -72,10 +74,8 @@ export interface ConfigSnapshot { data: Record; } -// ── In-memory store ────────────────────────────────────────────────────────── -// In production, persist to SQLite alongside other domain state. +// ── SQLite-backed store ─────────────────────────────────────────────────────── -let auditLog: ConfigAuditEntry[] = []; let idCounter = 0; function generateId(): string { @@ -85,6 +85,40 @@ function generateId(): string { return `audit-${ts}-${seq}`; } +function db() { + return getDbInstance(); +} + +interface ConfigAuditRow { + id: string; + timestamp: string; + action: string; + target: string; + target_id: string; + target_name: string; + before_json: string | null; + after_json: string | null; + diff_json: string; + source: string; + note: string | null; +} + +function rowToEntry(row: ConfigAuditRow): ConfigAuditEntry { + return { + id: row.id, + timestamp: row.timestamp, + action: row.action as AuditAction, + target: row.target as AuditTarget, + targetId: row.target_id, + targetName: row.target_name, + before: row.before_json === null ? null : (JSON.parse(row.before_json) as Record | null), + after: row.after_json === null ? null : (JSON.parse(row.after_json) as Record | null), + source: row.source as AuditSource, + diff: JSON.parse(row.diff_json) as ConfigDiff, + note: row.note, + }; +} + /** * Compute a structured diff between two configuration states. */ @@ -159,12 +193,24 @@ export function recordChange( note: note ?? null, }; - auditLog.push(entry); - - // Keep log bounded (max 1000 entries in memory) - if (auditLog.length > 1000) { - auditLog = auditLog.slice(-1000); - } + db().prepare( + `INSERT INTO config_audit_log + (id, timestamp, action, target, target_id, target_name, before_json, after_json, diff_json, source, note) + VALUES + (@id, @timestamp, @action, @target, @targetId, @targetName, @beforeJson, @afterJson, @diffJson, @source, @note)` + ).run({ + id: entry.id, + timestamp: entry.timestamp, + action: entry.action, + target: entry.target, + targetId: entry.targetId, + targetName: entry.targetName, + beforeJson: before === null ? null : JSON.stringify(before), + afterJson: after === null ? null : JSON.stringify(after), + diffJson: JSON.stringify(entry.diff), + source: entry.source, + note: entry.note, + }); return entry; } @@ -181,42 +227,57 @@ export function getAuditLog(options?: { limit?: number; offset?: number; }): { entries: ConfigAuditEntry[]; total: number } { - let filtered = auditLog; + const where: string[] = []; + const params: Record = {}; if (options?.target) { - filtered = filtered.filter((e) => e.target === options.target); + where.push("target = @target"); + params.target = options.target; } if (options?.targetId) { - filtered = filtered.filter((e) => e.targetId === options.targetId); + where.push("target_id = @targetId"); + params.targetId = options.targetId; } if (options?.action) { - filtered = filtered.filter((e) => e.action === options.action); + where.push("action = @action"); + params.action = options.action; } if (options?.source) { - filtered = filtered.filter((e) => e.source === options.source); + where.push("source = @source"); + params.source = options.source; } if (options?.since) { - filtered = filtered.filter((e) => e.timestamp >= options.since!); + where.push("timestamp >= @since"); + params.since = options.since; } - const total = filtered.length; + const whereSql = where.length > 0 ? `WHERE ${where.join(" AND ")}` : ""; - // Sort newest first - filtered = [...filtered].sort((a, b) => b.timestamp.localeCompare(a.timestamp)); + const totalRow = db() + .prepare(`SELECT COUNT(*) AS c FROM config_audit_log ${whereSql}`) + .get(params) as { c: number }; + const total = totalRow.c; - // Paginate const offset = options?.offset ?? 0; const limit = options?.limit ?? 50; - filtered = filtered.slice(offset, offset + limit); - return { entries: filtered, total }; + const rows = db() + .prepare( + `SELECT * FROM config_audit_log ${whereSql} ORDER BY datetime(timestamp) DESC, id DESC LIMIT @limit OFFSET @offset` + ) + .all({ ...params, limit, offset }) as ConfigAuditRow[]; + + return { entries: rows.map(rowToEntry), total }; } /** * Get a specific audit entry by ID. */ export function getAuditEntry(id: string): ConfigAuditEntry | null { - return auditLog.find((e) => e.id === id) ?? null; + const row = db() + .prepare("SELECT * FROM config_audit_log WHERE id = @id") + .get({ id }) as ConfigAuditRow | undefined; + return row ? rowToEntry(row) : null; } /** @@ -260,19 +321,23 @@ export function getAuditSummary(): { const byAction: Record = {}; const bySource: Record = {}; - for (const entry of auditLog) { - byTarget[entry.target] = (byTarget[entry.target] || 0) + 1; - byAction[entry.action] = (byAction[entry.action] || 0) + 1; - bySource[entry.source] = (bySource[entry.source] || 0) + 1; + const rows = db() + .prepare("SELECT * FROM config_audit_log ORDER BY datetime(timestamp) DESC, id DESC") + .all() as ConfigAuditRow[]; + + for (const row of rows) { + byTarget[row.target] = (byTarget[row.target] || 0) + 1; + byAction[row.action] = (byAction[row.action] || 0) + 1; + bySource[row.source] = (bySource[row.source] || 0) + 1; } return { - totalEntries: auditLog.length, + totalEntries: rows.length, byTarget, byAction, bySource, - oldestEntry: auditLog.length > 0 ? auditLog[0].timestamp : null, - newestEntry: auditLog.length > 0 ? auditLog[auditLog.length - 1].timestamp : null, + oldestEntry: rows.length > 0 ? rows[rows.length - 1].timestamp : null, + newestEntry: rows.length > 0 ? rows[0].timestamp : null, }; } @@ -280,6 +345,6 @@ export function getAuditSummary(): { * Reset the audit log. Useful for testing. */ export function resetAuditLog(): void { - auditLog = []; + db().prepare("DELETE FROM config_audit_log").run(); idCounter = 0; } diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index c7eeeb37542..1d1ae38830b 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -4974,7 +4974,9 @@ "skipped": "تم التكوين بالفعل", "failed": "فشل" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "المزودون", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "ربط IDE Cursor", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "الكشف التلقائي عن الرموز...", "readingFromCursor": "القراءة من Cursor IDE أو وكيل Cursor", "tokensAutoDetected": "تم اكتشاف الرموز المميزة تلقائيًا من Cursor IDE!", "cursorNotDetected": "لم يتم الكشف عن IDE للمؤشر. يرجى لصق الرمز المميز الخاص بك يدويًا.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "رمز الوصول", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "سيتم ملء رمز الوصول تلقائيًا...", "machineId": "معرف الآلة", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "غير قادر على اكتشاف الرموز المميزة تلقائيًا", "errorAutoDetectFailed": "فشل الكشف التلقائي عن الرموز المميزة", "errorEnterToken": "الرجاء إدخال رمز الوصول", - "errorImportFailed": "فشل الاستيراد" + "errorImportFailed": "فشل الاستيراد", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "تكوين التسعير", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "خطافات الويب", - "description": "تسجيل، وعرض، واختبار، وإزالة نقاط نهاية webhook. تكوين اشتراكات الأحداث (request.completed، وprovider.error، وbudget.exceeded، إلخ) وإدارة محاولات التسليم." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "خادم MCP", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 605fe9ed184..0a0254b70a3 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -4974,7 +4974,9 @@ "skipped": "artıq konfiqurasiya edilib", "failed": "başarısız oldu" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Vebhuklar", - "description": "Vebhuk son nöqtələrini qeydiyyatdan keçirin, siyahıya alın, sınaqdan keçirin və silin. Hadisə abunəliklərini (request.completed, provider.error, budget.exceeded və s.) konfiqurasiya edin və çatdırılmanın təkrar cəhdlərini idarə edin." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP Serveri", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index b2dfd1176a0..3450855c1c7 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -4974,7 +4974,9 @@ "skipped": "вече конфигурирано", "failed": "неуспешно" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Доставчици", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Свържете Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Автоматично откриване на токени...", "readingFromCursor": "Четене от Cursor IDE или cursor-agent", "tokensAutoDetected": "Токените са успешно автоматично открити от Cursor IDE!", "cursorNotDetected": "IDE на курсора не е открит. Моля, поставете ръчно вашето означение.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Токен за достъп", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Токенът за достъп ще се попълни автоматично...", "machineId": "ID на машината", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Не могат да се открият автоматично токени", "errorAutoDetectFailed": "Автоматичното откриване на токени не бе успешно", "errorEnterToken": "Моля, въведете токен за достъп", - "errorImportFailed": "Неуспешно импортиране" + "errorImportFailed": "Неуспешно импортиране", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Конфигурация на цените", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Уебхукове", - "description": "Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP сървър", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 34ecf02c74f..8d5a26b3ab2 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -4974,7 +4974,9 @@ "skipped": "আগে থেকেই কনফিগার করা আছে", "failed": "ব্যর্থ" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "ওয়েবহুক", - "description": "ওয়েবহুক এন্ডপয়েন্ট রেজিস্টার, তালিকাভুক্ত, পরীক্ষা এবং অপসারণ করুন। ইভেন্ট সাবস্ক্রিপশন (request.completed, provider.error, budget.exceeded ইত্যাদি) কনফিগার করুন এবং ডেলিভারি রিট্রাই পরিচালনা করুন।" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP সার্ভার", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 18761a6f44f..c35c6140c43 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -4974,7 +4974,9 @@ "skipped": "již nakonfigurováno", "failed": "neúspěšné" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Poskytovatelé", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Připojte kurzorové IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Automatická detekce tokenů...", "readingFromCursor": "Čtení z Cursor IDE nebo kurzorového agenta", "tokensAutoDetected": "Tokeny byly úspěšně automaticky detekovány z Cursor IDE!", "cursorNotDetected": "IDE kurzoru nebylo zjištěno. Vložte prosím svůj token ručně.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Přístupový token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Přístupový token se automaticky vyplní...", "machineId": "ID stroje", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Nelze automaticky detekovat tokeny", "errorAutoDetectFailed": "Automatická detekce tokenů se nezdařila", "errorEnterToken": "Zadejte přístupový token", - "errorImportFailed": "Import se nezdařil" + "errorImportFailed": "Import se nezdařil", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Konfigurace cen", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooky", - "description": "Registrujte, zobrazujte, testujte a odebírejte koncové body webhooků. Konfigurujte odběry událostí (request.completed, provider.error, budget.exceeded atd.) a spravujte opakování doručení." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP server", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index cd127ca8a33..764400b58b2 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -4974,7 +4974,9 @@ "skipped": "allerede konfigureret", "failed": "mislykkedes" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Udbydere", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Tilslut Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Automatisk registrering af tokens...", "readingFromCursor": "Læser fra Cursor IDE eller cursor-agent", "tokensAutoDetected": "Tokens blev automatisk registreret fra Cursor IDE!", "cursorNotDetected": "Markør-IDE blev ikke fundet. Indsæt venligst dit token manuelt.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Adgangstoken", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Adgangstoken udfyldes automatisk...", "machineId": "Maskin-id", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Kan ikke automatisk registrere tokens", "errorAutoDetectFailed": "Automatisk registrering af tokens mislykkedes", "errorEnterToken": "Indtast venligst adgangstoken", - "errorImportFailed": "Import mislykkedes" + "errorImportFailed": "Import mislykkedes", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Priskonfiguration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Registrer, list, test og fjern webhook-slutpunkter. Konfigurer begivenhedsabonnementer (request.completed, provider.error, budget.exceeded osv.) og administrer leveringsforsøg." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-server", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index d1f60bb1425..6a33f039f3c 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -4974,7 +4974,9 @@ "skipped": "bereits konfiguriert", "failed": "fehlgeschlagen" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Anbieter", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Verbinden Sie die Cursor-IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Automatische Erkennung von Token...", "readingFromCursor": "Lesen aus der Cursor-IDE oder dem Cursor-Agenten", "tokensAutoDetected": "Tokens wurden erfolgreich automatisch von der Cursor-IDE erkannt!", "cursorNotDetected": "Cursor-IDE nicht erkannt. Bitte fügen Sie Ihr Token manuell ein.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Zugriffstoken", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Das Zugriffstoken wird automatisch ausgefüllt...", "machineId": "Maschinen-ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Token können nicht automatisch erkannt werden", "errorAutoDetectFailed": "Die automatische Erkennung von Token ist fehlgeschlagen", "errorEnterToken": "Bitte geben Sie das Zugriffstoken ein", - "errorImportFailed": "Der Import ist fehlgeschlagen" + "errorImportFailed": "Der Import ist fehlgeschlagen", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Preiskonfiguration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Webhook-Endpunkte registrieren, auflisten, testen und entfernen. Ereignis-Abonnements (request.completed, provider.error, budget.exceeded etc.) konfigurieren und Zustellungsversuche verwalten." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-Server", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 2de1b5bd176..930f5b59587 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -4922,7 +4922,9 @@ "multiProvider": "Multi-Provider", "usageTracking": "Usage Tracking", "securityDesc": "Set a password to protect your dashboard, or skip for now.", + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", "providerDesc": "Connect your first AI provider. You can add more later.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard.", "apiKeyRequired": "API Key (required)", "customUrlOptional": "Custom URL (optional)", "testDesc": "Let's verify your provider connection works.", @@ -5690,8 +5692,8 @@ "aggregatorsGateways": "Aggregators Gateways", "enterpriseCloud": "Enterprise & Cloud", "apiFormatLabel": "Api Format Label", - "apiKeyOptionalHint": "Api Key Optional Hint", - "apiKeyOptionalLabel": "Api Key Optional Label", + "apiKeyOptionalHint": "Leave empty if your local setup or provider does not require authentication.", + "apiKeyOptionalLabel": "API Key (optional)", "apiRegionChina": "Api Region China", "apiRegionHint": "Api Region Hint", "apiRegionInternational": "Api Region International", @@ -6267,6 +6269,20 @@ "webSessionGuideStep3": "Copy the required credential from the provider's own domain. For cookies, copy only the Cookie header value and omit Cookie:.", "webSessionGuideStep3Manual": "Manual path: open the browser developer tools (F12 → Network), refresh the page, open an authenticated request, and copy the Cookie header value from Request Headers — omit the Cookie: prefix.", "webSessionGuideStep4": "Paste it here and check the connection. If it stops working, sign in again and replace it with a fresh value.", + "harImportButtonLabel": "Import .har file", + "harImportButtonBusy": "Importing…", + "harImportButtonHint": "Export from DevTools Network tab after sending at least one chat message.", + "harImportStatusValid": "Imported — valid for ~{minutes}m.", + "harImportStatusExpiringSoon": "Imported — valid for only ~{minutes}m more.", + "harImportStatusExpired": "Imported, but this token already expired ({minutes}m ago) — export a fresh HAR.", + "harImportStatusUnknownExpiry": "Imported. Couldn't read its expiry.", + "harImportErrorNotJson": "That file isn't valid JSON — is it really a .har export?", + "harImportErrorNoEntries": "This HAR has no network entries recorded.", + "harImportErrorNoChathubUrl": "No Copilot chat connection found in this HAR. Send at least one chat message in m365.cloud.microsoft before exporting.", + "harImportErrorUnparsableUrl": "Found the chat connection, but couldn't read its URL.", + "harImportErrorMissingFields": "Found the chat connection, but the token was missing from it.", + "harImportErrorReadFailed": "Couldn't read that file.", + "harImportErrorUnknown": "Couldn't extract a credential from that HAR file.", "webSessionSecurityHint": "Treat this like a password: it may access your signed-in web account until it expires or is revoked.", "webNoAuthGuideTitle": "No credential required", "webNoAuthGuideBody": "{provider} does not need an API key or cookie. Save the connection to use its free web endpoint.", @@ -10630,12 +10646,24 @@ "continue": "Continue" }, "cursorAuthModal": { - "title": "Connect Cursor IDE", + "title": "Connect Cursor", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", - "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "cursorNotDetected": "Cursor IDE not detected. Paste your access token (and refresh token if available).", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10647,7 +10675,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12190,7 +12221,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries." + "description": "Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP Server", @@ -13838,5 +13869,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "Cheaper Inference is an OmniRoute Open Source Friend", + "description": "A cost-ranked gateway reselling dozens of frontier models behind one OpenAI-compatible endpoint — routing each request to the cheapest eligible provider, never above list price.", + "cta": "Get an API Key", + "partnerLinkNote": "Partner link", + "dismissAriaLabel": "Dismiss" } } diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 94b17341c19..3ecd2c839cf 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -4974,7 +4974,9 @@ "skipped": "ya configurado", "failed": "fallido" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Proveedores", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Conectar cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Tokens de detección automática...", "readingFromCursor": "Lectura desde Cursor IDE o cursor-agent", "tokensAutoDetected": "¡Los tokens se detectaron automáticamente con éxito desde Cursor IDE!", "cursorNotDetected": "IDE del cursor no detectado. Pegue manualmente su token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Token de acceso", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "El token de acceso se completará automáticamente...", "machineId": "ID de máquina", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "No se pueden detectar tokens automáticamente", "errorAutoDetectFailed": "Error de detección automática de tokens", "errorEnterToken": "Por favor ingrese el token de acceso", - "errorImportFailed": "Importación fallida" + "errorImportFailed": "Importación fallida", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configuración de precios", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, provider.error, budget.exceeded, etc.) and manage delivery retries." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP Server", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 851f93aba97..4483a7ee845 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -4974,7 +4974,9 @@ "skipped": "قبلاً پیکربندی شده است", "failed": "ناموفق" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "وب‌هوک‌ها", - "description": "ثبت، فهرست کردن، آزمایش و حذف نقاط پایانی وب‌هوک. پیکربندی اشتراک‌های رویداد (request.completed، provider.error، budget.exceeded و غیره) و مدیریت تلاش‌های مجدد تحویل." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "سرور MCP", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 9149e2ec4fd..31f38e04a14 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -4974,7 +4974,9 @@ "skipped": "jo määritetty", "failed": "epäonnistui" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Palveluntarjoajat", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Yhdistä kohdistin IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Tunnistetaan automaattisesti...", "readingFromCursor": "Lukeminen Cursor IDE:stä tai cursor-agentista", "tokensAutoDetected": "Tokenit tunnistettiin automaattisesti Cursor IDE:stä!", "cursorNotDetected": "Kohdistimen IDE:tä ei havaittu. Liitä tunnus manuaalisesti.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Käyttöoikeustunnus", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Käyttöoikeustunnus täytetään automaattisesti...", "machineId": "Koneen tunnus", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Tunnuksia ei voi tunnistaa automaattisesti", "errorAutoDetectFailed": "Tunnusten automaattinen tunnistus epäonnistui", "errorEnterToken": "Anna käyttöoikeustunnus", - "errorImportFailed": "Tuonti epäonnistui" + "errorImportFailed": "Tuonti epäonnistui", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Hinnoitteluasetukset", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhookit", - "description": "Rekisteröi, listaa, testaa ja poista webhook-päätepisteitä. Määritä tapahtumatilaukset (request.completed, provider.error, budget.exceeded jne.) ja hallitse toimituksen uudelleenyrityksiä." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-palvelin", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 72a3792df2b..b065a34fee3 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -4974,7 +4974,9 @@ "skipped": "déjà configuré", "failed": "échec" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Fournisseurs", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connecter l'IDE du curseur", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Jetons à détection automatique...", "readingFromCursor": "Lecture à partir de Cursor IDE ou de Cursor-Agent", "tokensAutoDetected": "Jetons détectés automatiquement avec succès à partir de Cursor IDE !", "cursorNotDetected": "Curseur IDE non détecté. Veuillez coller manuellement votre jeton.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Jeton d'accès", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Le jeton d'accès se remplira automatiquement...", "machineId": "ID de l'ordinateur", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Impossible de détecter automatiquement les jetons", "errorAutoDetectFailed": "Échec de la détection automatique des jetons", "errorEnterToken": "Veuillez saisir le jeton d'accès", - "errorImportFailed": "Échec de l'importation" + "errorImportFailed": "Échec de l'importation", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configuration des prix", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Enregistrer, lister, tester et supprimer des points de terminaison de webhook. Configurer les abonnements aux événements (request.completed, provider.error, budget.exceeded, etc.) et gérer les tentatives de livraison." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Serveur MCP", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index a1a38f1980f..ce257984a0a 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -4974,7 +4974,9 @@ "skipped": "હવે જ કન્ફિગર કરેલ છે", "failed": "ફેલ થયું" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "વેબહૂક્સ", - "description": "વેબહૂક એન્ડપોઇન્ટ્સ રજીસ્ટર કરો, સૂચિબદ્ધ કરો, ટેસ્ટ કરો અને દૂર કરો. ઇવેન્ટ સબ્સ્ક્રિપ્શન્સ (request.completed, provider.error, budget.exceeded, વગેરે) કન્ફિગર કરો અને ડિલિવરી પુનઃપ્રયાસો મેનેજ કરો." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP સર્વર", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 96007496ae6..5281e5292c9 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -4974,7 +4974,9 @@ "skipped": "כבר מוגדר", "failed": "נכשל" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "ספקים", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "חבר את הסמן IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "מזהה אוטומטית אסימונים...", "readingFromCursor": "קריאה מ-IDE Cursor או Cursor-agent", "tokensAutoDetected": "אסימונים זוהו בהצלחה אוטומטית מ-Cursor IDE!", "cursorNotDetected": "IDE הסמן לא זוהה. אנא הדבק ידנית את האסימון שלך.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "אסימון גישה", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "אסימון הגישה יאוכלס אוטומטית...", "machineId": "מזהה מכונה", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "לא ניתן לזהות אוטומטית אסימונים", "errorAutoDetectFailed": "זיהוי אוטומטי של אסימונים נכשל", "errorEnterToken": "נא להזין אסימון גישה", - "errorImportFailed": "הייבוא נכשל" + "errorImportFailed": "הייבוא נכשל", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "תצורת תמחור", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "רשום, הצג, בדוק והסר נקודות קצה של webhook. הגדר מנויים לאירועים (request.completed, provider.error, budget.exceeded וכו') ונהל ניסיונות מסירה חוזרים." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "שרת MCP", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 3cbff8e2021..d74d2919ca9 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -4974,7 +4974,9 @@ "skipped": "पहले से कॉन्फ़िगर किया गया", "failed": "असफल" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "प्रदाता", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "कर्सर आईडीई कनेक्ट करें", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "टोकन का स्वतः पता लगाना...", "readingFromCursor": "कर्सर आईडीई या कर्सर-एजेंट से पढ़ना", "tokensAutoDetected": "कर्सर आईडीई से टोकन का सफलतापूर्वक स्वतः पता लगाया गया!", "cursorNotDetected": "कर्सर आईडीई का पता नहीं चला. कृपया अपना टोकन मैन्युअल रूप से चिपकाएँ।", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "प्रवेश टोकन", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "एक्सेस टोकन स्वतः-पॉप्युलेट हो जाएगा...", "machineId": "मशीन आईडी", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "टोकन का स्वतः पता लगाने में असमर्थ", "errorAutoDetectFailed": "टोकन का स्वतः पता लगाना विफल रहा", "errorEnterToken": "कृपया एक्सेस टोकन दर्ज करें", - "errorImportFailed": "आयात विफल" + "errorImportFailed": "आयात विफल", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "मूल्य निर्धारण विन्यास", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "वेबहुक", - "description": "वेबहुक एंडपॉइंट्स को रजिस्टर, सूचीबद्ध, टेस्ट और रिमूव करें। इवेंट सब्सक्रिप्शन (request.completed, provider.error, budget.exceeded, आदि) कॉन्फ़िगर करें और डिलीवरी रीट्राय प्रबंधित करें।" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP सर्वर", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 6f723ed8531..11d3327c80c 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -4974,7 +4974,9 @@ "skipped": "már konfigurálva van", "failed": "sikertelen" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Szolgáltatók", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Csatlakoztassa a kurzor IDE-t", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Tokenek automatikus felismerése...", "readingFromCursor": "Olvasás Cursor IDE-ből vagy cursor-agentből", "tokensAutoDetected": "A tokenek automatikusan felismerve a Cursor IDE-ből!", "cursorNotDetected": "A kurzor IDE nem észlelhető. Kérjük, manuálisan illessze be a tokent.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Hozzáférési token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "A hozzáférési token automatikusan kitöltődik...", "machineId": "Gépazonosító", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "A tokenek automatikus észlelése nem lehetséges", "errorAutoDetectFailed": "A tokenek automatikus észlelése sikertelen", "errorEnterToken": "Kérjük, adja meg a hozzáférési tokent", - "errorImportFailed": "Az importálás sikertelen" + "errorImportFailed": "Az importálás sikertelen", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Árképzési konfiguráció", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhookok", - "description": "Webhook-végpontok regisztrálása, listázása, tesztelése és eltávolítása. Eseményfeliratkozások (request.completed, provider.error, budget.exceeded stb.) konfigurálása és a kézbesítési újrapróbálkozások kezelése." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-szerver", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index f6aa07123ae..2bf5fdccdd9 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -4974,7 +4974,9 @@ "skipped": "sudah dikonfigurasi", "failed": "gagal" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Penyedia", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Hubungkan IDE Kursor", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Token yang terdeteksi secara otomatis...", "readingFromCursor": "Membaca dari IDE Kursor atau agen kursor", "tokensAutoDetected": "Token berhasil terdeteksi secara otomatis dari Cursor IDE!", "cursorNotDetected": "IDE kursor tidak terdeteksi. Harap tempelkan token Anda secara manual.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Akses Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Token akses akan terisi secara otomatis...", "machineId": "ID Mesin", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Tidak dapat mendeteksi token secara otomatis", "errorAutoDetectFailed": "Token deteksi otomatis gagal", "errorEnterToken": "Silakan masukkan token akses", - "errorImportFailed": "Impor gagal" + "errorImportFailed": "Impor gagal", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Konfigurasi Harga", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "Daftarkan, cantumkan, uji, dan hapus endpoint webhook. Konfigurasikan langganan peristiwa (request.completed, provider.error, budget.exceeded, dll.) dan kelola upaya pengiriman ulang." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Server MCP", diff --git a/src/i18n/messages/in.json b/src/i18n/messages/in.json index 5f2587eb6a8..afb01d57e8f 100644 --- a/src/i18n/messages/in.json +++ b/src/i18n/messages/in.json @@ -4974,7 +4974,9 @@ "skipped": "sudah dikonfigurasi", "failed": "gagal" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "Daftarkan, tampilkan daftar, uji, dan hapus endpoint webhook. Konfigurasikan langganan peristiwa (request.completed, provider.error, budget.exceeded, dll.) dan kelola percobaan ulang pengiriman." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Server MCP", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 902ed0795ad..dd69eeb4894 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -4974,7 +4974,9 @@ "skipped": "già configurato", "failed": "fallito" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Fornitori", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connetti l'IDE del cursore", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Token di rilevamento automatico...", "readingFromCursor": "Lettura da Cursor IDE o cursor-agent", "tokensAutoDetected": "Token rilevati automaticamente con successo da Cursor IDE!", "cursorNotDetected": "IDE del cursore non rilevato. Incolla manualmente il token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Gettone di accesso", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Il token di accesso si popolerà automaticamente...", "machineId": "ID macchina", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Impossibile rilevare automaticamente i token", "errorAutoDetectFailed": "Il rilevamento automatico dei token non è riuscito", "errorEnterToken": "Inserisci il token di accesso", - "errorImportFailed": "Importazione non riuscita" + "errorImportFailed": "Importazione non riuscita", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configurazione dei prezzi", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "Registra, elenca, testa e rimuovi gli endpoint webhook. Configura le sottoscrizioni agli eventi (request.completed, provider.error, budget.exceeded, ecc.) e gestisci i tentativi di recapito." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Server MCP", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index d586218bc86..371263c8eb1 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -4974,7 +4974,9 @@ "skipped": "すでに設定済み", "failed": "失敗しました" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "プロバイダー", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "カーソルIDEの接続", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "トークンを自動検出しています...", "readingFromCursor": "Cursor IDE またはカーソルエージェントからの読み取り", "tokensAutoDetected": "トークンは Cursor IDE から正常に自動検出されました。", "cursorNotDetected": "カーソル IDE が検出されません。トークンを手動で貼り付けてください。", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "アクセストークン", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "アクセストークンは自動入力されます...", "machineId": "マシンID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "トークンを自動検出できません", "errorAutoDetectFailed": "トークンの自動検出に失敗しました", "errorEnterToken": "アクセストークンを入力してください", - "errorImportFailed": "インポートに失敗しました" + "errorImportFailed": "インポートに失敗しました", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "価格設定", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "Webhookエンドポイントの登録、一覧表示、テスト、削除を行います。イベントサブスクリプション(request.completed、provider.error、budget.exceededなど)を設定し、配信の再試行を管理します。" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCPサーバー", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 4f10276f71b..fe5ba64093d 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -4974,7 +4974,9 @@ "skipped": "이미 구성됨", "failed": "실패했습니다" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "공급자", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Cursor IDE 연결", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "토큰 자동 감지 중...", "readingFromCursor": "Cursor IDE 또는 cursor-agent에서 읽는 중", "tokensAutoDetected": "Cursor IDE에서 토큰이 자동 감지되었습니다!", "cursorNotDetected": "Cursor IDE가 감지되지 않았습니다. 토큰을 수동으로 붙여넣으세요.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "액세스 토큰", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "액세스 토큰이 자동으로 채워집니다...", "machineId": "머신 ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "토큰을 자동 감지할 수 없습니다.", "errorAutoDetectFailed": "토큰 자동 감지 실패", "errorEnterToken": "액세스 토큰을 입력하세요.", - "errorImportFailed": "가져오기 실패" + "errorImportFailed": "가져오기 실패", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "가격 구성", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "웹훅", - "description": "웹훅 엔드포인트를 등록, 나열, 테스트 및 제거합니다. 이벤트 구독(request.completed, provider.error, budget.exceeded 등)을 구성하고 전달 재시도를 관리합니다." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP 서버", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 86437f4f3ac..487cd7ff2fc 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -4974,7 +4974,9 @@ "skipped": "आधीच कॉन्फिगर केले आहे", "failed": "अयशस्वी" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "वेबहूक्स", - "description": "वेबहूक एंडपॉइंट्स नोंदणीकृत करा, सूचीबद्ध करा, चाचणी करा आणि काढून टाका. इव्हेंट सबस्क्रिप्शन (request.completed, provider.error, budget.exceeded, इत्यादी) कॉन्फिगर करा आणि डिलिव्हरी रिट्राय व्यवस्थापित करा." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP सर्व्हर", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 498af0d138c..e317f33fdad 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -4974,7 +4974,9 @@ "skipped": "sudah dikonfigurasi", "failed": "gagal" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Pembekal", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Sambungkan IDE Kursor", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Token pengesan automatik...", "readingFromCursor": "Membaca daripada Cursor IDE atau cursor-agent", "tokensAutoDetected": "Token berjaya dikesan secara automatik daripada Cursor IDE!", "cursorNotDetected": "IDE kursor tidak dikesan. Sila tampal token anda secara manual.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Token Akses", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Token akses akan diisi secara automatik...", "machineId": "ID mesin", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Tidak dapat mengesan token secara automatik", "errorAutoDetectFailed": "Autokesan token gagal", "errorEnterToken": "Sila masukkan token akses", - "errorImportFailed": "Import gagal" + "errorImportFailed": "Import gagal", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Konfigurasi Harga", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "Daftar, senaraikan, uji dan alih keluar titik akhir webhook. Konfigurasikan langganan peristiwa (request.completed, provider.error, budget.exceeded, dll.) dan urus percubaan semula penghantaran." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Pelayan MCP", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 6422dad14d2..04e88d349cf 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -4974,7 +4974,9 @@ "skipped": "al reeds geconfigureerd", "failed": "mislukt" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Aanbieders", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Sluit Cursor-IDE aan", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Tokens automatisch detecteren...", "readingFromCursor": "Lezen vanuit Cursor IDE of cursor-agent", "tokensAutoDetected": "Tokens succesvol automatisch gedetecteerd vanuit Cursor IDE!", "cursorNotDetected": "Cursor-IDE niet gedetecteerd. Plak uw token handmatig.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Toegangstoken", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Toegangstoken wordt automatisch ingevuld...", "machineId": "Machine-ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Kan tokens niet automatisch detecteren", "errorAutoDetectFailed": "Automatische detectie van tokens is mislukt", "errorEnterToken": "Voer een toegangstoken in", - "errorImportFailed": "Importeren is mislukt" + "errorImportFailed": "Importeren is mislukt", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Prijsconfiguratie", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Registreer, toon, test en verwijder webhook-endpoints. Configureer gebeurtenisabonnementen (request.completed, provider.error, budget.exceeded, etc.) en beheer afleverpogingen." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-server", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 6f105c14c34..f07811a42be 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -4974,7 +4974,9 @@ "skipped": "allerede konfigurert", "failed": "mislyktes" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Leverandører", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Koble til markør-IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Automatisk oppdager tokens...", "readingFromCursor": "Leser fra Cursor IDE eller cursor-agent", "tokensAutoDetected": "Tokens ble automatisk oppdaget fra Cursor IDE!", "cursorNotDetected": "Markør-IDE ble ikke oppdaget. Vennligst lim inn tokenet ditt manuelt.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Tilgangstoken", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Tilgangstoken fylles ut automatisk...", "machineId": "Maskin-ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Kan ikke oppdage tokens automatisk", "errorAutoDetectFailed": "Automatisk gjenkjenning av tokens mislyktes", "errorEnterToken": "Vennligst skriv inn tilgangstoken", - "errorImportFailed": "Import mislyktes" + "errorImportFailed": "Import mislyktes", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Priskonfigurasjon", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Registrer, list opp, test og fjern webhook-endepunkter. Konfigurer hendelsesabonnementer (request.completed, provider.error, budget.exceeded osv.) og administrer leveringsforsøk." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-server", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index f9dc8a149b2..ba2811fe407 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -4974,7 +4974,9 @@ "skipped": "naka-configure na", "failed": "nabigo" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Mga provider", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Ikonekta ang Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Awtomatikong pagtukoy ng mga token...", "readingFromCursor": "Pagbabasa mula sa Cursor IDE o cursor-agent", "tokensAutoDetected": "Ang mga token ay matagumpay na na-auto-detect mula sa Cursor IDE!", "cursorNotDetected": "Hindi nakita ang cursor IDE. Mangyaring manu-manong i-paste ang iyong token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Awtomatikong magpo-populate ang token ng access...", "machineId": "ID ng makina", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Hindi ma-auto-detect ang mga token", "errorAutoDetectFailed": "Nabigo ang auto-detect na mga token", "errorEnterToken": "Mangyaring magpasok ng token ng pag-access", - "errorImportFailed": "Nabigo ang pag-import" + "errorImportFailed": "Nabigo ang pag-import", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configuration ng Pagpepresyo", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Mga Webhook", - "description": "Magrehistro, maglista, magsubok, at mag-alis ng mga webhook endpoint. I-configure ang mga subscription sa event (request.completed, provider.error, budget.exceeded, atbp.) at pamahalaan ang mga muling pagsubok sa paghahatid." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP Server", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 2bca5453b12..9eaba524665 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -4974,7 +4974,9 @@ "skipped": "już skonfigurowane", "failed": "niepowodzenie" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Połącz Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Automatyczne wykrywanie tokens...", "readingFromCursor": "Odczytywanie z Cursor IDE lub cursor-agent", "tokensAutoDetected": "Tokens pomyślnie automatycznie wykryte z Cursor IDE!", "cursorNotDetected": "Nie wykryto Cursor IDE. Wklej ręcznie swój token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token zostanie automatycznie uzupełniony...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Nie można automatycznie wykryć tokens", "errorAutoDetectFailed": "Automatyczne wykrywanie tokens nie powiodło się", "errorEnterToken": "Wprowadź access token", - "errorImportFailed": "Import nie powiódł się" + "errorImportFailed": "Import nie powiódł się", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Konfiguracja cennika", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooki", - "description": "Rejestruj, listuj, testuj i usuwaj punkty końcowe webhooków. Konfiguruj subskrypcje zdarzeń (request.completed, provider.error, budget.exceeded itp.) i zarządzaj ponownymi próbami dostarczenia." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Serwer MCP", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 9a3b93def69..f3e3c15d130 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -4979,7 +4979,9 @@ "skipped": "já configurado", "failed": "falhou" } - } + }, + "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você pode adicioná-los depois no painel após definir uma senha.", + "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois no painel." }, "providers": { "title": "Provedores", @@ -10631,11 +10633,23 @@ }, "cursorAuthModal": { "title": "Conecte o Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Detecção automática de tokens...", "readingFromCursor": "Lendo do Cursor IDE ou cursor-agent", "tokensAutoDetected": "Tokens detectados automaticamente com sucesso no Cursor IDE!", "cursorNotDetected": "Cursor IDE não detectado. Cole manualmente seu token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Token de acesso", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "O token de acesso será preenchido automaticamente...", "machineId": "ID da máquina", @@ -10647,7 +10661,10 @@ "errorAutoDetect": "Não foi possível detectar tokens automaticamente", "errorAutoDetectFailed": "Falha na detecção automática de tokens", "errorEnterToken": "Por favor insira o token de acesso", - "errorImportFailed": "Falha na importação" + "errorImportFailed": "Falha na importação", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configuração de preços", @@ -12190,7 +12207,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Registre, liste, teste e remova endpoints de webhook. Configure assinaturas de eventos (request.completed, provider.error, budget.exceeded, etc.) e gerencie novas tentativas de entrega." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index dc1a90c803c..1921f3fba7f 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -4974,7 +4974,9 @@ "skipped": "já configurado", "failed": "falhou" } - } + }, + "securityDescSkipWarning": "⚠️ Sem uma senha, não poderá adicionar fornecedores durante a configuração. Pode adicioná-los mais tarde no painel após definir uma senha.", + "providerRequiresPassword": "Precisa de definir uma senha primeiro para adicionar fornecedores. Volte ao passo de segurança e defina uma senha, ou adicione fornecedores mais tarde no painel." }, "providers": { "title": "Provedores", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Conecte o Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Detecção automática de tokens...", "readingFromCursor": "Lendo do Cursor IDE ou cursor-agent", "tokensAutoDetected": "Tokens detectados automaticamente com sucesso no Cursor IDE!", "cursorNotDetected": "Cursor IDE não detectado. Cole manualmente seu token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Token de acesso", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "O token de acesso será preenchido automaticamente...", "machineId": "ID da máquina", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Não foi possível detectar tokens automaticamente", "errorAutoDetectFailed": "Falha na detecção automática de tokens", "errorEnterToken": "Por favor insira o token de acesso", - "errorImportFailed": "Falha na importação" + "errorImportFailed": "Falha na importação", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configuração de preços", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Webhooks omni" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Servidor MCP", @@ -13828,5 +13845,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "A Cheaper Inference é uma Amiga do Código Aberto do OmniRoute", + "description": "Um gateway com custo ordenado que revende dezenas de modelos de fronteira num único endpoint compatível com OpenAI — roteando cada requisição ao provedor elegível mais barato, nunca acima do preço de tabela.", + "cta": "Obter uma Chave de API", + "partnerLinkNote": "Link de parceiro", + "dismissAriaLabel": "Dispensar" } } diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 4265215f2ee..4ae651a40a6 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -4974,7 +4974,9 @@ "skipped": "deja configurat", "failed": "eșuat" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Furnizorii", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Conectați Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Se detectează automat jetoanele...", "readingFromCursor": "Citirea din Cursor IDE sau cursor-agent", "tokensAutoDetected": "Jetoanele au fost detectate automat cu succes din Cursor IDE!", "cursorNotDetected": "IDE-ul cursorului nu a fost detectat. Vă rugăm să lipiți manual indicativul.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Token de acces", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Indicatorul de acces se va completa automat...", "machineId": "ID mașină", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Nu se pot detecta automat jetoanele", "errorAutoDetectFailed": "Jetoanele de detectare automată nu au reușit", "errorEnterToken": "Vă rugăm să introduceți simbolul de acces", - "errorImportFailed": "Importul nu a reușit" + "errorImportFailed": "Importul nu a reușit", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Configurarea prețurilor", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook-uri", - "description": "Înregistrează, listează, testează și elimină endpoint-uri de webhook. Configurează abonamentele la evenimente (request.completed, provider.error, budget.exceeded etc.) și gestionează reîncercările de livrare." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Server MCP", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 8210e133a26..94f6c651e05 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -4974,7 +4974,9 @@ "skipped": "уже настроено", "failed": "не удалось" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Провайдеры", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Подключить курсор IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Автоматическое обнаружение токенов...", "readingFromCursor": "Чтение из Cursor IDE или курсорного агента", "tokensAutoDetected": "Токены успешно автоматически обнаружены в Cursor IDE!", "cursorNotDetected": "Курсор IDE не обнаружен. Пожалуйста, вставьте свой токен вручную.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Токен доступа", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Токен доступа будет автоматически заполнен...", "machineId": "Идентификатор машины", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Невозможно автоматически обнаружить токены", "errorAutoDetectFailed": "Не удалось автоматически обнаружить токены", "errorEnterToken": "Пожалуйста, введите токен доступа", - "errorImportFailed": "Импорт не удался" + "errorImportFailed": "Импорт не удался", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Конфигурация цен", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Вебхуки", - "description": "Регистрация, просмотр списка, тестирование и удаление конечных точек вебхуков. Настройка подписок на события (request.completed, provider.error, budget.exceeded и т. д.) и управление повторными попытками доставки." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-сервер", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index ba4a09a0225..a6c823a524d 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -4974,7 +4974,9 @@ "skipped": "už nakonfigurované", "failed": "zlyhalo" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Poskytovatelia", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Pripojte kurzorové IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Automatické zisťovanie tokenov...", "readingFromCursor": "Čítanie z kurzorového IDE alebo kurzorového agenta", "tokensAutoDetected": "Tokeny boli úspešne automaticky zistené z Cursor IDE!", "cursorNotDetected": "Nebolo zistené IDE kurzora. Prilepte svoj token ručne.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Prístupový token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Prístupový token sa automaticky vyplní...", "machineId": "ID stroja", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Nie je možné automaticky rozpoznať tokeny", "errorAutoDetectFailed": "Automatická detekcia tokenov zlyhala", "errorEnterToken": "Zadajte prístupový token", - "errorImportFailed": "Import zlyhal" + "errorImportFailed": "Import zlyhal", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Konfigurácia cien", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooky", - "description": "Registrujte, zobrazujte, testujte a odstraňujte koncové body webhookov. Konfigurujte odbery udalostí (request.completed, provider.error, budget.exceeded atď.) a spravujte opakované pokusy o doručenie." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP Server", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 1433b5b49ce..dd61134b1c6 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -4974,7 +4974,9 @@ "skipped": "redan konfigurerad", "failed": "misslyckades" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Leverantörer", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Anslut Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Identifierar tokens automatiskt...", "readingFromCursor": "Läser från Cursor IDE eller cursor-agent", "tokensAutoDetected": "Tokens har framgångsrikt identifierats automatiskt från Cursor IDE!", "cursorNotDetected": "Markör-IDE upptäcktes inte. Vänligen klistra in din token manuellt.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Åtkomsttoken kommer att fyllas i automatiskt...", "machineId": "Maskin-ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Det går inte att automatiskt upptäcka tokens", "errorAutoDetectFailed": "Automatisk identifiering av tokens misslyckades", "errorEnterToken": "Vänligen ange åtkomsttoken", - "errorImportFailed": "Importen misslyckades" + "errorImportFailed": "Importen misslyckades", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Priskonfiguration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Registrera, lista, testa och ta bort webhook-slutpunkter. Konfigurera händelseprenumerationer (request.completed, provider.error, budget.exceeded osv.) och hantera återförsök vid leverans." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP-server", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 3a2b3c204ca..7042ca73e99 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -4974,7 +4974,9 @@ "skipped": "tayari imewekwa", "failed": "imefeli" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "Sajili, orodhesha, jaribu, na uondoe vituo vya mwisho vya webhook. Sanidi usajili wa matukio (request.completed, provider.error, budget.exceeded, n.k.) na udhibiti majaribio tena ya uwasilishaji." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Seva ya MCP", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index ab9fc3c1f32..28eef20efcb 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -4974,7 +4974,9 @@ "skipped": "முன்னதாக கட்டமைக்கப்பட்டுள்ளது", "failed": "தோல்வி" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "வெப்ஹூக்குகள்", - "description": "வெப்ஹூக் எண்ட்பாயிண்ட்டுகளைப் பதிவு செய்யவும், பட்டியலிடவும், சோதிக்கவும் மற்றும் அகற்றவும். நிகழ்வு சந்தாக்களை (request.completed, provider.error, budget.exceeded போன்றவை) கட்டமைக்கவும் மற்றும் விநியோக மறுமுயற்சிகளை நிர்வகிக்கவும்." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP சர்வர்", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 6d1a07e41b3..acecbfb4035 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -4974,7 +4974,9 @@ "skipped": "ఇప్పటికే కాన్ఫిగర్ చేయబడింది", "failed": "ఫెయిల్డ్" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "వెబ్‌హుక్స్", - "description": "Webhook ఎండ్‌పాయింట్‌లను నమోదు చేయండి, జాబితా చేయండి, పరీక్షించండి మరియు తీసివేయండి. ఈవెంట్ సబ్‌స్క్రిప్షన్‌లను (request.completed, provider.error, budget.exceeded, మొదలైనవి) కాన్ఫిగర్ చేయండి మరియు డెలివరీ రీట్రైలను నిర్వహించండి." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP సర్వర్", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index b30a5bebc5e..9fb3122583f 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -4974,7 +4974,9 @@ "skipped": "กำหนดค่าเรียบร้อยแล้ว", "failed": "ล้มเหลว" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "ผู้ให้บริการ", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "เชื่อมต่อเคอร์เซอร์ IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "กำลังตรวจจับโทเค็นอัตโนมัติ...", "readingFromCursor": "อ่านจากเคอร์เซอร์ IDE หรือเคอร์เซอร์เอเจนต์", "tokensAutoDetected": "โทเค็นตรวจพบอัตโนมัติสำเร็จจาก Cursor IDE!", "cursorNotDetected": "ตรวจไม่พบเคอร์เซอร์ IDE โปรดวางโทเค็นของคุณด้วยตนเอง", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "โทเค็นการเข้าถึง", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "โทเค็นการเข้าถึงจะเติมข้อมูลอัตโนมัติ...", "machineId": "หมายเลขเครื่อง", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "ไม่สามารถตรวจจับโทเค็นอัตโนมัติได้", "errorAutoDetectFailed": "โทเค็นการตรวจจับอัตโนมัติล้มเหลว", "errorEnterToken": "กรุณาใส่โทเค็นการเข้าถึง", - "errorImportFailed": "การนำเข้าล้มเหลว" + "errorImportFailed": "การนำเข้าล้มเหลว", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "การกำหนดค่าราคา", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "เว็บฮุค", - "description": "ลงทะเบียน, แสดงรายการ, ทดสอบ และลบปลายทางเว็บฮุค กำหนดค่าการสมัครรับเหตุการณ์ (request.completed, provider.error, budget.exceeded ฯลฯ) และจัดการการพยายามส่งใหม่" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "เซิร์ฟเวอร์ MCP", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 958b931040a..c05cb34b228 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -4974,7 +4974,9 @@ "skipped": "zaten yapılandırılmış", "failed": "başarısız" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Sağlayıcılar", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "İmleç IDE'sini bağlayın", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Belirteçler otomatik olarak algılanıyor...", "readingFromCursor": "İmleç IDE'sinden veya imleç aracısından okuma", "tokensAutoDetected": "Belirteçler Cursor IDE'den başarıyla otomatik olarak algılandı!", "cursorNotDetected": "İmleç IDE'si algılanmadı. Lütfen jetonunuzu manuel olarak yapıştırın.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Erişim Jetonu", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Erişim belirteci otomatik olarak doldurulacak...", "machineId": "Makine Kimliği", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Belirteçler otomatik olarak algılanamıyor", "errorAutoDetectFailed": "Belirteçlerin otomatik algılanması başarısız oldu", "errorEnterToken": "Lütfen erişim belirtecini girin", - "errorImportFailed": "İçe aktarma başarısız oldu" + "errorImportFailed": "İçe aktarma başarısız oldu", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Fiyatlandırma Yapılandırması", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook'lar", - "description": "Webhook uç noktalarını kaydedin, listeleyin, test edin ve kaldırın. Olay aboneliklerini (request.completed, provider.error, budget.exceeded vb.) yapılandırın ve teslimat yeniden denemelerini yönetin." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP Sunucusu", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index 586571a9ca3..7eb5a5e5678 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -4974,7 +4974,9 @@ "skipped": "вже налаштовано", "failed": "не вдалося" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Провайдери", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Підключіть Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Автоматичне визначення токенів...", "readingFromCursor": "Читання з Cursor IDE або cursor-agent", "tokensAutoDetected": "Маркери успішно автоматично виявлені в Cursor IDE!", "cursorNotDetected": "Курсор IDE не виявлено. Вставте маркер вручну.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Маркер доступу", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Маркер доступу буде заповнено автоматично...", "machineId": "ID машини", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Неможливо автоматично визначити маркери", "errorAutoDetectFailed": "Помилка автоматичного визначення маркерів", "errorEnterToken": "Будь ласка, введіть маркер доступу", - "errorImportFailed": "Помилка імпорту" + "errorImportFailed": "Помилка імпорту", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Конфігурація ціни", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Вебхуки", - "description": "Реєструйте, переглядайте, тестуйте та видаляйте кінцеві точки вебхуків. Налаштовуйте підписки на події (request.completed, provider.error, budget.exceeded тощо) та керуйте повторними спробами доставки." + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "Сервер MCP", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index aef5f7ae73e..5d5e2b1c84b 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -4974,7 +4974,9 @@ "skipped": "پہلے سے ترتیب دیا گیا", "failed": "ناکام" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "Providers", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "Connect Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Auto-detecting tokens...", "readingFromCursor": "Reading from Cursor IDE or cursor-agent", "tokensAutoDetected": "Tokens successfully auto-detected from Cursor IDE!", "cursorNotDetected": "Cursor IDE not detected. Please manually paste your token.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Access Token", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Access token will auto-populate...", "machineId": "Machine ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "Unable to auto-detect tokens", "errorAutoDetectFailed": "Auto-detect tokens failed", "errorEnterToken": "Please enter access token", - "errorImportFailed": "Import failed" + "errorImportFailed": "Import failed", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Pricing Configuration", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "ویب ہکس", - "description": "ویب ہک اینڈ پوائنٹس کو رجسٹر کریں، فہرست بنائیں، ٹیسٹ کریں اور ہٹائیں۔ ایونٹ سبسکرپشنز (request.completed، provider.error، budget.exceeded، وغیرہ) کو ترتیب دیں اور ڈیلیوری کی دوبارہ کوششوں کا نظم کریں۔" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP سرور", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 0004b404b9a..e04b2066613 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -4979,7 +4979,9 @@ "skipped": "đã được cấu hình", "failed": "thất bại" } - } + }, + "securityDescSkipWarning": "⚠️ Không có mật khẩu, bạn sẽ không thể thêm nhà cung cấp trong quá trình thiết lập. Bạn có thể thêm sau từ bảng điều khiển sau khi đặt mật khẩu.", + "providerRequiresPassword": "Bạn cần đặt mật khẩu trước khi thêm nhà cung cấp. Hãy quay lại bước bảo mật và đặt mật khẩu, hoặc thêm nhà cung cấp sau từ bảng điều khiển." }, "providers": { "title": "Nhà cung cấp", @@ -6267,6 +6269,20 @@ "webSessionGuideStep3": "Sao chép thông tin xác thực được yêu cầu từ tên miền riêng của nhà cung cấp. Đối với cookie, chỉ sao chép giá trị tiêu đề Cookie và bỏ qua Cookie:.", "webSessionGuideStep3Manual": "Cách thủ công: mở công cụ dành cho nhà phát triển của trình duyệt (F12 → Network), tải lại trang, mở một yêu cầu đã xác thực và sao chép giá trị tiêu đề Cookie trong Request Headers — bỏ tiền tố Cookie:.", "webSessionGuideStep4": "Dán vào đây và kiểm tra kết nối. Nếu nó ngừng hoạt động, hãy đăng nhập lại và thay thế bằng một giá trị mới.", + "harImportButtonLabel": "Nhập tệp .har", + "harImportButtonBusy": "Đang nhập…", + "harImportButtonHint": "Xuất từ thẻ Network của DevTools sau khi gửi ít nhất một tin nhắn chat.", + "harImportStatusValid": "Đã nhập — hợp lệ trong ~{minutes} phút.", + "harImportStatusExpiringSoon": "Đã nhập — chỉ còn hợp lệ ~{minutes} phút nữa.", + "harImportStatusExpired": "Đã nhập, nhưng token này đã hết hạn ({minutes} phút trước) — hãy xuất một HAR mới.", + "harImportStatusUnknownExpiry": "Đã nhập. Không đọc được thời hạn.", + "harImportErrorNotJson": "Tệp đó không phải JSON hợp lệ — có đúng là bản xuất .har không?", + "harImportErrorNoEntries": "HAR này không có mục network nào được ghi lại.", + "harImportErrorNoChathubUrl": "Không tìm thấy kết nối Copilot chat trong HAR này. Hãy gửi ít nhất một tin nhắn chat trong m365.cloud.microsoft trước khi xuất.", + "harImportErrorUnparsableUrl": "Tìm thấy kết nối chat, nhưng không đọc được URL của nó.", + "harImportErrorMissingFields": "Tìm thấy kết nối chat, nhưng token bị thiếu trong đó.", + "harImportErrorReadFailed": "Không đọc được tệp đó.", + "harImportErrorUnknown": "Không trích xuất được thông tin xác thực từ tệp HAR đó.", "webSessionSecurityHint": "Hãy coi đây như mật khẩu: nó có thể truy cập vào tài khoản web đã đăng nhập của bạn cho đến khi hết hạn hoặc bị thu hồi.", "webNoAuthGuideTitle": "Không yêu cầu thông tin xác thực", "webNoAuthGuideBody": "{provider} không cần khóa API hoặc cookie. Lưu kết nối để sử dụng endpoint web miễn phí của nó.", @@ -10631,11 +10647,23 @@ }, "cursorAuthModal": { "title": "Kết nối Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "Tự động phát hiện token...", "readingFromCursor": "Đang đọc token từ Cursor IDE hoặc cursor-agent", "tokensAutoDetected": "Đã tự động phát hiện token từ Cursor IDE thành công!", "cursorNotDetected": "Không phát hiện thấy Cursor IDE. Vui lòng dán token của bạn thủ công.", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "Token truy cập", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "Token truy cập sẽ được tự động điền...", "machineId": "Mã máy", @@ -10647,7 +10675,10 @@ "errorAutoDetect": "Không thể tự động phát hiện token", "errorAutoDetectFailed": "Tự động phát hiện token không thành công", "errorEnterToken": "Vui lòng nhập token truy cập", - "errorImportFailed": "Nhập thất bại" + "errorImportFailed": "Nhập thất bại", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "Cấu hình giá", @@ -12190,7 +12221,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "Đăng ký, liệt kê, kiểm tra và xóa endpoint webhook. Cấu hình đăng ký sự kiện như request.completed, provider.error, budget.exceeded và quản lý thử gửi lại." + "description": "Đăng ký, liệt kê, kiểm thử và xoá các endpoint webhook. Cấu hình đăng ký sự kiện (request.completed, request.failed, quota.exceeded, v.v.) và quản lý thử lại giao hàng." }, "omni-mcp": { "name": "Máy chủ MCP", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 05eed0cd999..7a96cd92b33 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -4974,7 +4974,9 @@ "skipped": "已配置", "failed": "失败" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "提供者", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "连接 Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "自动检测令牌中...", "readingFromCursor": "正在从 Cursor IDE 或 cursor-agent 读取", "tokensAutoDetected": "已成功从 Cursor IDE 自动检测到令牌!", "cursorNotDetected": "未检测到 Cursor IDE。请手动粘贴您的令牌。", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "访问令牌", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "访问令牌将自动填充...", "machineId": "机器 ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "无法自动检测令牌", "errorAutoDetectFailed": "自动检测令牌失败", "errorEnterToken": "请输入访问令牌", - "errorImportFailed": "导入失败" + "errorImportFailed": "导入失败", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "定价配置", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhooks", - "description": "注册、列出、测试和移除 Webhook 端点。配置事件订阅(request.completed、provider.error、budget.exceeded 等)并管理投递重试。" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP 服务器", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 442b9a66e74..101fdbb9489 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -4974,7 +4974,9 @@ "skipped": "已配置", "failed": "失敗" } - } + }, + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard." }, "providers": { "title": "提供者", @@ -10626,11 +10628,23 @@ }, "cursorAuthModal": { "title": "連線 Cursor IDE", + "tabLogin": "Login with Cursor", + "tabImport": "Import token", + "loginDescription": "Opens Cursor's browser login. Works in Docker — approve in your host browser, then return here.", + "loginWithCursor": "Login with Cursor", + "startingLogin": "Starting…", + "waitingApproval": "Waiting for Cursor login approval…", + "openUrlHint": "If the browser did not open:", + "openLoginLink": "Open login page", + "cancelLogin": "Cancel login", "autoDetecting": "自動檢測權杖中...", "readingFromCursor": "正在從 Cursor IDE 或 cursor-agent 讀取", "tokensAutoDetected": "已成功從 Cursor IDE 自動檢測到權杖!", "cursorNotDetected": "未檢測到 Cursor IDE。請手動貼上您的權杖。", + "dockerImportHint": "Running in Docker? Prefer Login with Cursor. IDE auto-import and cursor-agent are usually unavailable inside the container. See docs/providers/CURSOR-DOCKER.md.", "accessToken": "訪問權杖", + "refreshToken": "Refresh Token", + "refreshTokenPlaceholder": "Optional — enables automatic token refresh", "required": "*", "accessTokenPlaceholder": "訪問權杖將自動填充...", "machineId": "機器 ID", @@ -10642,7 +10656,10 @@ "errorAutoDetect": "無法自動檢測權杖", "errorAutoDetectFailed": "自動檢測權杖失敗", "errorEnterToken": "請輸入訪問權杖", - "errorImportFailed": "匯入失敗" + "errorImportFailed": "匯入失敗", + "errorLoginStart": "Failed to start Cursor login", + "errorLoginPoll": "Cursor login failed", + "errorLoginTimeout": "Cursor login timed out — try again" }, "pricingModal": { "title": "定價設定", @@ -12185,7 +12202,7 @@ }, "omni-webhooks": { "name": "Webhook", - "description": "註冊、列出、測試和移除 Webhook 端點。設定事件訂閱(request.completed、provider.error、budget.exceeded 等)並管理傳遞重試。" + "description": "__MISSING__:Register, list, test, and remove webhook endpoints. Configure event subscriptions (request.completed, request.failed, quota.exceeded, etc.) and manage delivery retries." }, "omni-mcp": { "name": "MCP 伺服器", diff --git a/src/instrumentation-node.ts b/src/instrumentation-node.ts index fec6a90329f..c622f4cfa71 100755 --- a/src/instrumentation-node.ts +++ b/src/instrumentation-node.ts @@ -471,8 +471,12 @@ export async function registerNodejs(): Promise { // instrumentation startup), NOT in the unused src/server-init.ts. try { const { initCredentialHealthCheck } = await import("@/lib/credentialHealth/scheduler"); - initCredentialHealthCheck(); - console.log("[STARTUP] Credential health scheduler started"); + const started = initCredentialHealthCheck(); + console.log( + started + ? "[STARTUP] Credential health scheduler started" + : "[STARTUP] Credential health scheduler disabled" + ); } catch (err: unknown) { const msg = err instanceof Error ? err.message : String(err); console.warn("[STARTUP] Could not start credential health scheduler:", msg); diff --git a/src/lib/acp/registry.ts b/src/lib/acp/registry.ts index 93315a546df..07408f62360 100644 --- a/src/lib/acp/registry.ts +++ b/src/lib/acp/registry.ts @@ -200,6 +200,14 @@ let _customAgentDefs: CustomAgentDef[] = []; const DISALLOWED_VERSION_COMMAND_CHARS = /[;&|<>`$\r\n]/; +// A version probe only ever needs a version flag. For untrusted (client-registered) +// custom agents the binary-match check alone is not enough: the caller controls both +// `binary` and `versionCommand`, so a matching interpreter with an eval-style argument +// (`node -e …`, `python -c …`, `ruby -e …`) reaches execFileSync as arbitrary code +// execution without any shell metacharacter. Restricting the args to a recognized +// version flag closes that path — see GHSA-jphr-2gw7-xrwp / GHSA-hf57-cqmx-p4gr. +const SAFE_VERSION_PROBE_ARG = /^(-v|-V|--version|-version|version|--ver)$/; + /** * Set custom agent definitions from settings. */ @@ -300,6 +308,12 @@ export function resolveVersionProbe( if (!allowed.has(normalizedCommand)) { return null; } + + // Untrusted probe: allow only a bare binary or a single recognized version + // flag, so a matching interpreter cannot smuggle an eval/exec argument. + if (args.length > 1 || (args.length === 1 && !SAFE_VERSION_PROBE_ARG.test(args[0]))) { + return null; + } } return { command, args }; diff --git a/src/lib/cli-helper/config-generator/opencode.ts b/src/lib/cli-helper/config-generator/opencode.ts index c2915e5eb7f..e65206dc557 100644 --- a/src/lib/cli-helper/config-generator/opencode.ts +++ b/src/lib/cli-helper/config-generator/opencode.ts @@ -243,9 +243,11 @@ function resolveContextLength(entry: CatalogModelEntry): number | undefined { * 1. Existing manual override in the user's opencode.json (`limit.context`). * 2. Catalog `context_length` / `max_context_window_tokens`. * - * If neither is available, the entry is returned WITHOUT a `limit` block so - * the caller can decide whether to skip the model entirely or surface a - * warning. We never fabricate a default context window. + * If neither is available, `limit.context` is simply omitted and OpenCode's + * own heuristics apply — we never fabricate a default context window. The + * entry ALWAYS carries a `limit` block, though: `limit.output` is a + * required field in OpenCode's v1 provider schema, so it is always emitted + * (falling back to 8K when nothing else is known) — see #10940. */ function buildModelEntry( id: string, @@ -278,10 +280,9 @@ function buildModelEntry( } // Resolve the context window. Honor an explicit user override, then fall - // back to the catalog. We do NOT synthesize a default — if the catalog - // is unaware of a model's window, the opencode.json will simply omit - // `limit.context` for that model and OpenCode's own heuristics apply. - // (OpenCode v1 defaults to 128K when `limit.context` is missing.) + // back to the catalog. If the catalog is unaware of a model's window, we + // fall back to a safe default (128K) so OpenCode's v1 provider schema + // validator never rejects the config with a missing key error (#11035). const userLimit = existing?.limit?.context; const catalogLimit = catalog ? resolveContextLength(catalog) : undefined; const context = typeof userLimit === "number" && userLimit > 0 ? userLimit : catalogLimit; @@ -290,9 +291,6 @@ function buildModelEntry( // Use the catalog's max_output_tokens when available; otherwise fall // back to the user's existing `limit.output` and finally to a small // default (8K) so OpenCode never errors on a totally missing output cap. - // We do NOT default context — context is a property of the model and - // we have no business guessing. Output is a per-request setting and a - // small default is harmless when truly unknown. const userOutput = existing?.limit?.output; const catalogOutput = catalog && typeof catalog.max_output_tokens === "number" && catalog.max_output_tokens > 0 @@ -301,25 +299,26 @@ function buildModelEntry( const output = typeof userOutput === "number" && userOutput > 0 ? userOutput : (catalogOutput ?? 8_192); - // Emit `limit` only if we have at least one of context/output. We never - // emit a half-baked limit block with only an `output` (would be misleading). - if ( - typeof context === "number" || - typeof userOutput === "number" || - typeof catalogOutput === "number" - ) { - const limit: { context?: number; input?: number; output?: number } = {}; - if (typeof context === "number") limit.context = context; - limit.output = output; - const userInput = existing?.limit?.input; - if (typeof userInput === "number" && userInput > 0) { - limit.input = userInput; - } else if (catalog) { - const maxInput = catalog.max_input_tokens; - if (typeof maxInput === "number" && maxInput > 0) limit.input = maxInput; - } - entry.limit = limit; + // Both `limit.context` and `limit.output` are REQUIRED by OpenCode's v1 provider schema + // regardless of whether the catalog (or the user's existing config) knows the model's + // context window — a model with no catalog metadata at all must still get both + // `limit.context` and `limit.output`, or OpenCode rejects the whole config with "Missing key + // provider.omniroute.models.{model}.limit.context" (#11035) or ".limit.output" (#10940, #11032). + // `output` above resolves to a safe fallback (8K) and `context` resolves to a safe fallback (128K) + // when nothing else is known, so we always emit both fields. + const resolvedContext = typeof context === "number" && context > 0 ? context : 128_000; + const limit: { context: number; input?: number; output: number } = { + context: resolvedContext, + output, + }; + const userInput = existing?.limit?.input; + if (typeof userInput === "number" && userInput > 0) { + limit.input = userInput; + } else if (catalog) { + const maxInput = catalog.max_input_tokens; + if (typeof maxInput === "number" && maxInput > 0) limit.input = maxInput; } + entry.limit = limit; return entry; } @@ -408,7 +407,7 @@ export interface GenerateOpencodeOptions { /** * If `true` (default), the generator fetches the live `/v1/models` catalog * so every model entry has an explicit `limit.context`. The catalog is the - * single source of truth for context windows; we never invent defaults. + * primary source of truth for context windows, falling back to 128K when unknown. * * When the catalog request fails, the generator throws — opencode.json must * not be emitted with stale or fabricated values. The CLI can catch the @@ -423,7 +422,7 @@ export interface GenerateOpencodeOptions { /** * Generate a full `opencode.json` document for OmniRoute. The catalog is the - * single source of truth for context windows — we never hardcode values. + * primary source of truth for context windows, with a 128K fallback when unknown. * * Behavior: * - Preserves the user's existing provider name, npm, options, and @@ -433,8 +432,8 @@ export interface GenerateOpencodeOptions { * - For each catalog model id the user did NOT have, a new entry is * added with `limit.context` populated when the catalog has it. * - If the catalog has no context for a model AND the user has no - * override, the model is emitted WITHOUT a `limit.context` field. - * OpenCode's own heuristic (typically 128K) applies. + * override, a safe default (128K) is emitted so OpenCode's schema validator + * does not reject the model. * - Throws if the catalog fetch fails — the user must fix the upstream * before we can generate a reliable opencode.json. */ diff --git a/src/lib/config/runtimeSettings.ts b/src/lib/config/runtimeSettings.ts index dc5e9d720dc..bf615749bae 100644 --- a/src/lib/config/runtimeSettings.ts +++ b/src/lib/config/runtimeSettings.ts @@ -1,5 +1,9 @@ import { clearHealthCheckLogCache } from "@/lib/tokenHealthCheck"; import { setCustomBannedSignals } from "@omniroute/open-sse/services/accountFallback.ts"; +import { + setOperatorProviderErrorRules, + type OperatorProviderErrorRule, +} from "@omniroute/open-sse/config/providerErrorRules.ts"; import { isAutomatedTestProcess } from "@/shared/utils/testProcess"; type JsonRecord = Record; @@ -46,6 +50,7 @@ interface RuntimeSettingsSnapshot { systemTransforms: unknown; authzBypass: AuthzBypassSnapshot; customBannedSignals: string[]; + providerErrorRules: Record | null; } // Default bypass policy: kill-switch on, `/api/mcp/` bypassable. Mirrors the @@ -72,6 +77,7 @@ const DEFAULT_RUNTIME_SETTINGS_SNAPSHOT: RuntimeSettingsSnapshot = { systemTransforms: null, authzBypass: DEFAULT_AUTHZ_BYPASS_SNAPSHOT, customBannedSignals: [], + providerErrorRules: null, }; let lastAppliedSnapshot: RuntimeSettingsSnapshot | null = null; @@ -138,6 +144,34 @@ function normalizeStringArray(value: unknown): string[] { ); } +/** + * Defensive shape-check of operator-declared error rules pulled from settings. + * The settings schema already validates this on write; this guard prevents a + * malformed stored value (or an unexpected shape) from crashing the + * error-classification hot path. Returns null when the value is missing or not + * a record of non-empty rule arrays. + */ +function normalizeOperatorProviderErrorRules( + value: unknown +): Record | null { + if (value === null || typeof value !== "object") return null; + const record = value as Record; + const result: Record = {}; + for (const [provider, list] of Object.entries(record)) { + if (!Array.isArray(list) || list.length === 0) continue; + const rules = list.filter( + (entry): entry is OperatorProviderErrorRule => + !!entry && + typeof entry === "object" && + typeof (entry as OperatorProviderErrorRule).status === "number" && + typeof (entry as OperatorProviderErrorRule).match === "string" && + typeof (entry as OperatorProviderErrorRule).scope === "string" + ); + if (rules.length > 0) result[provider.toLowerCase()] = rules; + } + return Object.keys(result).length > 0 ? result : null; +} + function normalizeStringRecord(value: unknown): Record { const record = toRecord(parseStoredJson(value, "modelAliases")); const entries = Object.entries(record) @@ -244,6 +278,7 @@ export function buildRuntimeSettingsSnapshot( systemTransforms: parseStoredJson(settings.systemTransforms, "systemTransforms"), authzBypass: normalizeAuthzBypass(settings), customBannedSignals: normalizeStringArray(settings.customBannedSignals), + providerErrorRules: normalizeOperatorProviderErrorRules(settings.providerErrorRules), }; } @@ -540,6 +575,13 @@ export async function applyRuntimeSettings( markChanged("bannedSignals"); } + if ( + force || + hasChanged(currentSnapshot.providerErrorRules, previousSnapshot.providerErrorRules) + ) { + setOperatorProviderErrorRules(currentSnapshot.providerErrorRules ?? undefined); + } + lastAppliedSnapshot = currentSnapshot; return changes; } diff --git a/src/lib/credentialHealth/scheduler.ts b/src/lib/credentialHealth/scheduler.ts index a407f6a326a..62b2d41f4a8 100644 --- a/src/lib/credentialHealth/scheduler.ts +++ b/src/lib/credentialHealth/scheduler.ts @@ -349,10 +349,13 @@ function scheduleSweep(): void { /** * Start the credential health check scheduler (idempotent). + * Returns whether the sweep is armed. False when + * OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK is set (#11016). */ -export function initCredentialHealthCheck(): void { +export function initCredentialHealthCheck(): boolean { const state = getSchedulerState(); - if (state.initialized || isCredentialHealthCheckDisabled()) return; + if (isCredentialHealthCheckDisabled()) return false; + if (state.initialized) return true; state.initialized = true; initCredentialCache(); @@ -364,6 +367,7 @@ export function initCredentialHealthCheck(): void { state.sweepTimer = setTimeout(() => { sweep().catch((err) => console.error(LOG_PREFIX, "Initial sweep failed:", err)); }, INITIAL_DELAY_MS); + return true; } /** diff --git a/src/lib/cursor/tokenExtractor.ts b/src/lib/cursor/tokenExtractor.ts index fda379004b5..1ca27939a21 100644 --- a/src/lib/cursor/tokenExtractor.ts +++ b/src/lib/cursor/tokenExtractor.ts @@ -64,6 +64,7 @@ export async function verifyLinuxCursorInstalled(probe: CursorInstallProbe = {}) * exact match wins. */ const ACCESS_TOKEN_KEYS = ["cursorAuth/accessToken", "cursorAuth/token"] as const; +const REFRESH_TOKEN_KEYS = ["cursorAuth/refreshToken"] as const; const MACHINE_ID_KEYS = [ "storage.serviceMachineId", "storage.machineId", @@ -92,12 +93,13 @@ interface VscDbRow { interface ExtractedCursorTokens { accessToken?: string; + refreshToken?: string; machineId?: string; } /** - * Pick the first matching access-token / machine-id from a set of rows. - * Pure function — easy to unit-test without a SQLite handle. + * Pick the first matching access-token / refresh-token / machine-id from a + * set of rows. Pure function — easy to unit-test without a SQLite handle. */ export function extractCursorTokensFromRows(rows: VscDbRow[]): ExtractedCursorTokens { const tokens: ExtractedCursorTokens = {}; @@ -105,6 +107,12 @@ export function extractCursorTokensFromRows(rows: VscDbRow[]): ExtractedCursorTo if (!tokens.accessToken && (ACCESS_TOKEN_KEYS as readonly string[]).includes(row.key)) { const v = normalizeVscDbValue(row.value); if (typeof v === "string") tokens.accessToken = v; + } else if ( + !tokens.refreshToken && + (REFRESH_TOKEN_KEYS as readonly string[]).includes(row.key) + ) { + const v = normalizeVscDbValue(row.value); + if (typeof v === "string") tokens.refreshToken = v; } else if (!tokens.machineId && (MACHINE_ID_KEYS as readonly string[]).includes(row.key)) { const v = normalizeVscDbValue(row.value); if (typeof v === "string") tokens.machineId = v; @@ -114,10 +122,11 @@ export function extractCursorTokensFromRows(rows: VscDbRow[]): ExtractedCursorTo } /** - * Fuzzy-match access-token / machine-id from any rows whose key vaguely - * resembles the expected pattern (e.g. `cursorAuth/someOtherAccessTokenKey`, - * `storage.someMachineId`). Used only when the exact-key lookup yielded - * nothing — guards against silent breakage when Cursor renames a key. + * Fuzzy-match access-token / refresh-token / machine-id from any rows whose + * key vaguely resembles the expected pattern (e.g. + * `cursorAuth/someOtherAccessTokenKey`, `storage.someMachineId`). Used only + * when the exact-key lookup yielded nothing — guards against silent breakage + * when Cursor renames a key. */ export function fuzzyExtractCursorTokensFromRows( rows: VscDbRow[], @@ -130,6 +139,9 @@ export function fuzzyExtractCursorTokensFromRows( const value = normalizeVscDbValue(row.value); if (typeof value !== "string") continue; if (!tokens.accessToken && lower.includes("accesstoken")) tokens.accessToken = value; + if (!tokens.refreshToken && lower.includes("refreshtoken") && !lower.includes("accesstoken")) { + tokens.refreshToken = value; + } if (!tokens.machineId && lower.includes("machineid")) tokens.machineId = value; } return tokens; @@ -245,6 +257,7 @@ export async function tryAgentAuth(): Promise<{ export async function tryIdeAuth(options?: { timeoutMs?: number }): Promise<{ found: boolean; accessToken?: string; + refreshToken?: string; machineId?: string; source?: string; error?: string; @@ -329,7 +342,7 @@ export async function tryIdeAuth(options?: { timeoutMs?: number }): Promise<{ } try { - const desiredKeys = [...ACCESS_TOKEN_KEYS, ...MACHINE_ID_KEYS]; + const desiredKeys = [...ACCESS_TOKEN_KEYS, ...REFRESH_TOKEN_KEYS, ...MACHINE_ID_KEYS]; const placeholders = desiredKeys.map(() => "?").join(","); const rows = db .prepare(`SELECT key, value FROM itemTable WHERE key IN (${placeholders})`) @@ -360,6 +373,7 @@ export async function tryIdeAuth(options?: { timeoutMs?: number }): Promise<{ return { found: true, accessToken: tokens.accessToken, + refreshToken: tokens.refreshToken, machineId: tokens.machineId, source: "cursor-ide", }; diff --git a/src/lib/db/adapters/driverFactory.ts b/src/lib/db/adapters/driverFactory.ts index 134e984c28f..304b334bd8e 100644 --- a/src/lib/db/adapters/driverFactory.ts +++ b/src/lib/db/adapters/driverFactory.ts @@ -212,8 +212,7 @@ export function createSyncDriverFactory(load: DriverLoader, betterSqliteProbe?: filePath: string, options?: Record ): SqliteAdapter | null { - // Bun ships a supported SQLite implementation. Prefer it over the native - // Node addon, which Bun intentionally skips because its ABI is incompatible. + // 1. Bun native sqlite driver: preferred built-in driver when running under Bun if (process.versions.bun) { try { const { Database } = load("bun:sqlite") as { @@ -222,18 +221,20 @@ export function createSyncDriverFactory(load: DriverLoader, betterSqliteProbe?: if (options?.fileMustExist === true && filePath !== ":memory:" && !existsSync(filePath)) { throw new Error(`SQLite file does not exist: ${filePath}`); } - const db = new Database(filePath, { - ...(options?.readonly === true - ? { readonly: true } - : { readwrite: true, create: options?.fileMustExist !== true }), - }); + const bunOptions: Record = {}; + if (options?.readonly === true) bunOptions.readonly = true; + if (options?.create === false && filePath !== ":memory:") bunOptions.create = false; + const db = + Object.keys(bunOptions).length > 0 + ? new Database(filePath, bunOptions) + : new Database(filePath); return createBunSqliteAdapter(db, filePath); } catch (err) { logSwallowedDriverError("bun:sqlite", err); } } - // better-sqlite3: rápido, nativo — skip em Bun + // 2. better-sqlite3: preferred native driver on Node.js if (!process.versions.bun && mayLoadBetterSqlite()) { try { const BetterSqlite = load("better-sqlite3") as { @@ -242,7 +243,6 @@ export function createSyncDriverFactory(load: DriverLoader, betterSqliteProbe?: const db = new BetterSqlite(filePath, options); return createBetterSqliteAdapter(db); } catch (err) { - // continua para próximo driver logSwallowedDriverError("better-sqlite3", err); } } diff --git a/src/lib/db/callLogStats.ts b/src/lib/db/callLogStats.ts index 7df9908bd17..f3845f31f15 100644 --- a/src/lib/db/callLogStats.ts +++ b/src/lib/db/callLogStats.ts @@ -25,6 +25,15 @@ export interface ProviderMetricRow { lastErrorStatus: number | null; } +/** One provider's traffic over a bounded window. See `getProviderUsageSince`. */ +export interface ProviderUsageRow { + provider: string; + requests: number; + successes: number; + avgLatencyMs: number | null; + lastRequestAt: string | null; +} + export interface SearchProviderStatRow { provider: string; requests: number; @@ -107,6 +116,43 @@ export function getProviderMetrics(): ProviderMetricRow[] { .all() as ProviderMetricRow[]; } +// --------------------------------------------------------------------------- +// /api/free-provider-rankings — windowed usage aggregate +// --------------------------------------------------------------------------- + +/** + * Per-provider usage over a time window: how much traffic a provider actually + * served, and how much of it succeeded. + * + * Deliberately NOT `getProviderMetrics()` with a `since` parameter: that query + * carries two correlated subqueries (`lastStatus`, `lastErrorStatus`) which a + * ranking never displays, and they dominate its cost — `call_logs` is indexed + * on `timestamp` alone, so each correlated pass rescans the whole window per + * provider. Here a single bounded `GROUP BY` uses `idx_cl_timestamp` and stops + * there. The rules are shared with its neighbour, not the query: same success + * definition, same `#10714` guard against providers whose connections are gone. + */ +export function getProviderUsageSince(since: string): ProviderUsageRow[] { + const db = getDbInstance(); + return db + .prepare( + `SELECT + c.provider, + COUNT(*) as requests, + SUM(CASE WHEN c.status >= 200 AND c.status < 400 THEN 1 ELSE 0 END) as successes, + ROUND(AVG(c.duration)) as avgLatencyMs, + MAX(c.timestamp) as lastRequestAt + FROM call_logs c + WHERE c.provider IS NOT NULL AND c.provider != '-' + AND c.timestamp >= @since + AND EXISTS ( + SELECT 1 FROM provider_connections pc WHERE pc.provider = c.provider + ) + GROUP BY c.provider` + ) + .all({ since }) as ProviderUsageRow[]; +} + // --------------------------------------------------------------------------- // /api/search/stats — search provider aggregates + recent entries // --------------------------------------------------------------------------- diff --git a/src/lib/db/cleanup.ts b/src/lib/db/cleanup.ts index 617bf4c2264..60e288bf4b4 100644 --- a/src/lib/db/cleanup.ts +++ b/src/lib/db/cleanup.ts @@ -193,6 +193,31 @@ export async function cleanupMcpAudit(): Promise { return result; } +/** + * Clean up old config_audit_log based on retention settings. + */ +export async function cleanupConfigAudit(retentionDays = getRetentionSettings().configAudit): Promise { + const db = getDbInstance(); + const result: CleanupResult = { deleted: 0, errors: 0 }; + + try { + const stmt = db.prepare( + "DELETE FROM config_audit_log WHERE datetime(timestamp) < datetime('now', '-' || ? || ' days')" + ); + const runResult = stmt.run(String(retentionDays)); + result.deleted = runResult.changes; + + console.log( + `[Cleanup] Deleted ${result.deleted} config_audit_log older than ${retentionDays} days` + ); + } catch (err: unknown) { + console.error("[Cleanup] Error cleaning config_audit_log:", err); + result.errors++; + } + + return result; +} + /** * Clean up old a2a_task_events based on retention settings. */ @@ -420,6 +445,7 @@ export async function runAutoCleanup(): Promise<{ usageHistory: await cleanupUsageHistory(), compressionAnalytics: await cleanupCompressionAnalytics(), mcpAudit: await cleanupMcpAudit(), + configAudit: await cleanupConfigAudit(), a2aEvents: await cleanupA2aEvents(), memoryEntries: await cleanupMemoryEntries(), domainCostHistory: await cleanupDomainCostHistory(), diff --git a/src/lib/db/databaseSettings.ts b/src/lib/db/databaseSettings.ts index e18f2a66bd9..0a12729242c 100644 --- a/src/lib/db/databaseSettings.ts +++ b/src/lib/db/databaseSettings.ts @@ -46,6 +46,7 @@ const LEGACY_FLAT_KEYS: { quotaSnapshots: ["quotaSnapshots"], compressionAnalytics: ["compressionAnalytics"], mcpAudit: ["mcpAudit"], + configAudit: ["configAudit"], a2aEvents: ["a2aEvents"], callLogs: ["callLogs"], usageHistory: ["usageHistory"], diff --git a/src/lib/db/gamification.ts b/src/lib/db/gamification.ts index 731c5d82adf..cd6c907170f 100644 --- a/src/lib/db/gamification.ts +++ b/src/lib/db/gamification.ts @@ -131,13 +131,24 @@ export function getRank(apiKeyId: string, scope: string): number { return rankRow.rank; } +export const LEADERBOARD_MAX_LIMIT = 200; + export function getTopN(scope: string, limit: number, offset: number = 0): LeaderboardRow[] { + // Guard the SQLite LIMIT/OFFSET bind. A negative LIMIT means "no limit" in + // SQLite (returns the whole table), and a non-integer throws a datatype + // mismatch, so an unvalidated `?limit` on a caller route (e.g. the leaderboard + // endpoints) could read the entire leaderboard or 500. Clamp to a coherent + // range here as a defense-in-depth backstop, independent of route validation. + const safeLimit = Number.isFinite(limit) + ? Math.min(Math.max(Math.trunc(limit), 0), LEADERBOARD_MAX_LIMIT) + : 0; + const safeOffset = Number.isFinite(offset) ? Math.max(Math.trunc(offset), 0) : 0; const rows = db() .prepare( `SELECT api_key_id, scope, score, updated_at FROM leaderboard WHERE scope = ? ORDER BY score DESC LIMIT ? OFFSET ?` ) - .all(scope, limit, offset) as Array<{ + .all(scope, safeLimit, safeOffset) as Array<{ api_key_id: string; scope: string; score: number; diff --git a/src/lib/db/healthCheck.ts b/src/lib/db/healthCheck.ts index 39165800361..cb46d28e082 100644 --- a/src/lib/db/healthCheck.ts +++ b/src/lib/db/healthCheck.ts @@ -48,6 +48,53 @@ export function describeDbDriver(db: Pick): Db }; } +export interface PagerCorruptionNote { + source: string; + message: string; + at: string; +} + +let pagerCorruption: PagerCorruptionNote | null = null; + +function pagerErrorMessage(error: unknown): string { + if (error instanceof Error) return error.message; + if (error && typeof error === "object" && "message" in error) { + return String((error as { message?: unknown }).message ?? ""); + } + return String(error ?? "unknown"); +} + +export function isSqlitePagerCorruptError(error: unknown): boolean { + const code = + error && typeof error === "object" && "code" in error + ? String((error as { code?: unknown }).code || "") + : ""; + const message = pagerErrorMessage(error); + return ( + code === "SQLITE_CORRUPT" || + code === "SQLITE_NOTADB" || + code === "SQLITE_IOERR" || + /malformed|SQLITE_CORRUPT|SQLITE_NOTADB|SQLITE_IOERR/i.test(message) + ); +} + +export function notePagerCorruption(source: string, error: unknown): void { + const message = pagerErrorMessage(error) || "unknown"; + pagerCorruption = { + source, + message, + at: new Date().toISOString(), + }; +} + +export function getPagerCorruption(): PagerCorruptionNote | null { + return pagerCorruption; +} + +export function resetPagerCorruption(): void { + pagerCorruption = null; +} + interface RunDbHealthCheckOptions { autoRepair?: boolean; createBackupBeforeRepair?: () => boolean; @@ -425,6 +472,14 @@ export function runDbHealthCheck( const expectedSchemaVersion = options.expectedSchemaVersion || "1"; const checkedAt = new Date().toISOString(); const issues: DbHealthIssue[] = []; + if (pagerCorruption) { + issues.push({ + type: "integrity_check_failed", + table: "sqlite", + description: `Pager reported SQLITE_CORRUPT during ${pagerCorruption.source}: ${pagerCorruption.message}`, + count: 1, + }); + } let repairedCount = 0; let backupCreated = false; let backupAttempted = false; diff --git a/src/lib/db/migrations/046_database_settings.sql b/src/lib/db/migrations/046_database_settings.sql index 57fd15c903b..6fb9864390a 100644 --- a/src/lib/db/migrations/046_database_settings.sql +++ b/src/lib/db/migrations/046_database_settings.sql @@ -25,6 +25,7 @@ INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSetting INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'quotaSnapshots', '90'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'compressionAnalytics', '30'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'mcpAudit', '30'); +INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'configAudit', '30'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'a2aEvents', '30'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'callLogs', '90'); INSERT OR IGNORE INTO key_value (namespace, key, value) VALUES ('databaseSettings', 'usageHistory', '365'); diff --git a/src/lib/db/migrations/161_config_audit_log.sql b/src/lib/db/migrations/161_config_audit_log.sql new file mode 100644 index 00000000000..0aad91ace28 --- /dev/null +++ b/src/lib/db/migrations/161_config_audit_log.sql @@ -0,0 +1,15 @@ +CREATE TABLE IF NOT EXISTS config_audit_log ( + id TEXT PRIMARY KEY, + timestamp TEXT NOT NULL, + action TEXT NOT NULL, + target TEXT NOT NULL, + target_id TEXT NOT NULL, + target_name TEXT NOT NULL, + before_json TEXT, + after_json TEXT, + diff_json TEXT NOT NULL, + source TEXT NOT NULL, + note TEXT +); +CREATE INDEX IF NOT EXISTS idx_config_audit_log_target_created ON config_audit_log(target, timestamp); +CREATE INDEX IF NOT EXISTS idx_config_audit_log_created ON config_audit_log(timestamp); diff --git a/src/lib/db/migrations/162_remove_hackclub_provider.sql b/src/lib/db/migrations/162_remove_hackclub_provider.sql new file mode 100644 index 00000000000..4dd20904e96 --- /dev/null +++ b/src/lib/db/migrations/162_remove_hackclub_provider.sql @@ -0,0 +1,21 @@ +-- 162_remove_hackclub_provider.sql +-- Hack Club AI provider was removed from OmniRoute at the request of Hack Club's +-- maintainers (#11118). Clean up any locally stored configuration for it. +-- Historical request and usage records are intentionally preserved under the +-- provider identity that existed when they were written. + +DELETE FROM provider_connections +WHERE provider = 'hackclub'; + +DELETE FROM registered_keys +WHERE provider = 'hackclub'; + +DELETE FROM provider_key_limits +WHERE provider = 'hackclub'; + +DELETE FROM discovery_results +WHERE provider_id = 'hackclub'; + +DELETE FROM key_value +WHERE namespace = 'customModels' + AND key = 'hackclub'; diff --git a/src/lib/db/models/syncedAvailableModelPersistence.ts b/src/lib/db/models/syncedAvailableModelPersistence.ts index 7e8d13a1429..5f7257ff2c3 100644 --- a/src/lib/db/models/syncedAvailableModelPersistence.ts +++ b/src/lib/db/models/syncedAvailableModelPersistence.ts @@ -7,6 +7,33 @@ import { getKeyValue } from "./shared"; type ModelNormalizer = (models: unknown) => T[]; +// #11016: the Feature 5004 reconciler copies provider-declared windows +// (`inputTokenLimit`, captured at /models discovery) into `auto:discovery` +// overrides — the only source the REQUEST-TIME token-limit chain trusts for a +// freshly synced model that models.dev has not indexed yet. Before this, the +// reconcile ran only at startup + every 24h, so a model synced mid-cycle was +// enforced at the provider's static `defaultContextLength` (128K for +// OpenRouter) for up to a day even though the catalog advertised its real +// window (measured: `openrouter/stealth/ox-alpha` advertised 1,048,576, +// enforced 128,000). Run the reconcile opportunistically right after a synced +// catalog write changes, debounced + fire-and-forget so the sync request is +// never blocked (dynamic import: `contextWindowResolver` reads this module via +// `getAllSyncedAvailableModels`, so a static import would be circular). +let reconcileAfterSyncTimer: ReturnType | null = null; + +function scheduleReconcileAfterSyncWrite(): void { + if (reconcileAfterSyncTimer) return; // debounce bursty multi-connection writes + reconcileAfterSyncTimer = setTimeout(() => { + reconcileAfterSyncTimer = null; + void import("../../contextWindowResolver") + .then((m) => m.runContextWindowReconcile()) + .catch(() => { + // Swallow — the periodic reconcile still runs; sync must never fail on it. + }); + }, 0); + reconcileAfterSyncTimer.unref?.(); +} + export function finishSyncedAvailableModelsWrite(): void { backupDbFile("pre-write"); invalidateModelCatalogCache(); @@ -48,5 +75,6 @@ export function persistCanonicalSyncedAvailableModels( ).run(key, JSON.stringify(normalizedModels)); } finishSyncedAvailableModelsWrite(); + scheduleReconcileAfterSyncWrite(); return true; } diff --git a/src/lib/db/providers.ts b/src/lib/db/providers.ts index 3b02933e113..fbab6817fd6 100644 --- a/src/lib/db/providers.ts +++ b/src/lib/db/providers.ts @@ -627,15 +627,26 @@ export async function createProviderConnection(data: JsonRecord) { // to no-overrides) keeps the field present on the returned object so the // UI can tell "field was read, no overrides" apart from "field absent." if ("quotaWindowThresholds" in connection) { - connection.quotaWindowThresholds = sanitizeQuotaWindowThresholds( - connection.quotaWindowThresholds - ); + const result = sanitizeQuotaWindowThresholds(connection.quotaWindowThresholds); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist quotaWindowThresholds with rejected keys: ${result.rejected.join(", ")}` + ); + } + connection.quotaWindowThresholds = result.sanitized; } // Same sanitization for rateLimitOverrides — keep in-memory representation - // in sync with what gets persisted. + // in sync with what gets persisted. Reject (don't silently drop) invalid + // keys/values so a direct DB writer can't lose operator intent. if ("rateLimitOverrides" in connection) { - connection.rateLimitOverrides = sanitizeRateLimitOverrides(connection.rateLimitOverrides); + const result = sanitizeRateLimitOverrides(connection.rateLimitOverrides); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist rateLimitOverrides with rejected keys: ${result.rejected.join(", ")}` + ); + } + connection.rateLimitOverrides = result.sanitized; } _insertConnectionRow(db, encryptConnectionFields({ ...connection })); @@ -849,13 +860,24 @@ export async function updateProviderConnection(id: string, data: JsonRecord) { // Mirror the sanitization the create path applies — keep the returned // object in lockstep with what we persist. if ("quotaWindowThresholds" in merged) { - const sanitized = sanitizeQuotaWindowThresholds(merged.quotaWindowThresholds); + const result = sanitizeQuotaWindowThresholds(merged.quotaWindowThresholds); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist quotaWindowThresholds with rejected keys: ${result.rejected.join(", ")}` + ); + } // For updates we always carry the key forward (even as null) so the read - // path surfaces the cleared state to callers that just patched it. - merged.quotaWindowThresholds = sanitized; + // path surfaces the cleared state to callers that merged it. + merged.quotaWindowThresholds = result.sanitized; } if ("rateLimitOverrides" in merged) { - merged.rateLimitOverrides = sanitizeRateLimitOverrides(merged.rateLimitOverrides); + const result = sanitizeRateLimitOverrides(merged.rateLimitOverrides); + if (result.rejected.length > 0) { + throw new Error( + `Refusing to persist rateLimitOverrides with rejected keys: ${result.rejected.join(", ")}` + ); + } + merged.rateLimitOverrides = result.sanitized; } const existingRecord = toRecord(existing); diff --git a/src/lib/db/providers/columns.ts b/src/lib/db/providers/columns.ts index b32f653f7c6..f06f948cd21 100644 --- a/src/lib/db/providers/columns.ts +++ b/src/lib/db/providers/columns.ts @@ -64,20 +64,37 @@ export function normalizeBooleanColumn(value: unknown, fallback: boolean): boole return fallback; } +// Result of sanitizing a per-connection overrides/threshold map. `sanitized` +// is the cleaned value (or null when it collapses to nothing); `rejected` +// lists every key that was refused so callers can fail loudly +// instead of silently dropping the operator's input. +export type SanitizeResult = { + sanitized: Record | null; + rejected: string[]; +}; + // Sanitize the per-connection rate limit overrides map: keep only known -// fields with valid numeric values. Called once at each write-path boundary. -export function sanitizeRateLimitOverrides(value: unknown): Record | null { - if (value === null || value === undefined) return null; - if (typeof value !== "object" || Array.isArray(value)) return null; +// fields with valid non-negative integer values. Called once at each +// write-path boundary. Unknown keys and invalid values go into `rejected` +// rather than being dropped in silence. +export function sanitizeRateLimitOverrides(value: unknown): SanitizeResult { + if (value === null || value === undefined) return { sanitized: null, rejected: [] }; + if (typeof value !== "object" || Array.isArray(value)) return { sanitized: null, rejected: [] }; const allowedKeys = new Set(["rpm", "tpm", "tpd", "minTime", "maxConcurrent"]); + const rejected: string[] = []; const map: Record = {}; for (const [key, v] of Object.entries(value as Record)) { - if (!allowedKeys.has(key)) continue; + if (!allowedKeys.has(key)) { + rejected.push(key); + continue; + } if (typeof v === "number" && Number.isInteger(v) && v >= 0) { map[key] = v; + } else { + rejected.push(key); } } - return Object.keys(map).length === 0 ? null : map; + return { sanitized: Object.keys(map).length === 0 ? null : map, rejected }; } // Serialize an already-sanitized map for SQLite TEXT storage. @@ -91,20 +108,29 @@ export function toRecord(value: unknown): JsonRecord { return value && typeof value === "object" ? (value as JsonRecord) : {}; } -// Sanitize the per-window threshold map: keep only 0-100 integer values. -// Called once at each write-path boundary (createProviderConnection + -// updateProviderConnection) so both the in-memory return and the persisted -// row share the same shape. Serialization below trusts this output. -export function sanitizeQuotaWindowThresholds(value: unknown): Record | null { - if (value === null || value === undefined) return null; - if (typeof value !== "object" || Array.isArray(value)) return null; +// Sanitize the per-window threshold map: keep only 0-100 integer values with +// keys no longer than 64 chars. Called once at each write-path boundary +// (createProviderConnection + updateProviderConnection) so both the in-memory +// return and the persisted row share the same shape. Serialization below +// trusts this output. Invalid keys/values go into `rejected` rather than being +// dropped in silence. +export function sanitizeQuotaWindowThresholds(value: unknown): SanitizeResult { + if (value === null || value === undefined) return { sanitized: null, rejected: [] }; + if (typeof value !== "object" || Array.isArray(value)) return { sanitized: null, rejected: [] }; + const rejected: string[] = []; const map: Record = {}; for (const [key, v] of Object.entries(value as Record)) { + if (key.length > 64) { + rejected.push(key); + continue; + } if (typeof v === "number" && Number.isInteger(v) && v >= 0 && v <= 100) { map[key] = v; + } else { + rejected.push(key); } } - return Object.keys(map).length === 0 ? null : map; + return { sanitized: Object.keys(map).length === 0 ? null : map, rejected }; } export function toStringOrNull(value: unknown): string | null { diff --git a/src/lib/db/settings.ts b/src/lib/db/settings.ts index ce2b8890661..b9393db3c8f 100644 --- a/src/lib/db/settings.ts +++ b/src/lib/db/settings.ts @@ -238,6 +238,11 @@ export async function getSettings() { // (`:free` suffix, zero-price pricing, or FREE_MODEL_BUDGETS membership). Default // false preserves prior behaviour; opt-in only. hidePaidModels: false, + // Opt-in, default off: same shape as hidePaidModels above, but requires a + // live hard-stop-guaranteed quota check for non-keyless free candidates. + // See open-sse/services/autoCombo/strictZeroCostFilter.ts. + freeAccessPolicy: "off", + excludeTosAvoid: false, // #9418: Opt-in filter that hides auto/* virtual combos from the /v1/models catalog. // User-defined combos are unaffected; routing still works for hidden ids sent explicitly. hideAutoCombos: false, diff --git a/src/lib/embeddings/service.ts b/src/lib/embeddings/service.ts index e7337f49e2a..3341de28993 100644 --- a/src/lib/embeddings/service.ts +++ b/src/lib/embeddings/service.ts @@ -243,6 +243,12 @@ export async function createEmbeddingResponse( credentials.retryAfterHuman ); } + if ("allExpired" in credentials && credentials.allExpired) { + return errorResponse( + HTTP_STATUS.UNAUTHORIZED, + `[${provider}] All ${credentials.expiredCount || 1} connection(s) authentication expired — please reconnect in the dashboard` + ); + } } else if (provider === "ollama-local") { // Ollama is keyless, but a configured connection can still provide a // custom local host. Hydrate that optional connection without imposing an @@ -323,7 +329,7 @@ export async function createEmbeddingResponse( const responseHeaders = new Headers(result.headers); if (result.success) { - if (credentials) await clearRecoveredProviderState(credentials); + if (credentials) await clearRecoveredProviderState(credentials as Record); responseHeaders.set("Content-Type", "application/json"); const usage = (result.data as { usage?: Record })?.usage ?? null; const costUsd = usage ? await calculateCost(provider, effectiveModel ?? "", usage) : 0; diff --git a/src/lib/freeProviderRankings.ts b/src/lib/freeProviderRankings.ts index a0fa1ecd07d..37c99095c63 100644 --- a/src/lib/freeProviderRankings.ts +++ b/src/lib/freeProviderRankings.ts @@ -12,9 +12,14 @@ import { NOAUTH_PROVIDERS, OAUTH_PROVIDERS, APIKEY_PROVIDERS } from "@/shared/co import { REGISTRY } from "@omniroute/open-sse/config/providerRegistry"; import { listModelIntelligence } from "./db/modelIntelligence"; import { getProviderConnections } from "./db/providers"; +import { getProviderUsageSince, type ProviderUsageRow } from "./db/callLogStats"; import { getCustomModels } from "./db/models"; // Type-only: reuse the health vocabulary instead of forking it. -import type { ProviderHealthState } from "./monitoring/providerHealthMatrix"; +import { RANGE_MS } from "./monitoring/providerHealthMatrix"; +import type { + ProviderHealthState, + ProviderHealthMatrixRange, +} from "./monitoring/providerHealthMatrix"; import type { ProviderAuthType } from "./freeProviderRankingsAuthType"; // Re-exported for backward-compat / same-module ergonomics (#6915) — the @@ -248,8 +253,31 @@ export interface ProviderReliability { }>; /** Provider aggregate; absent entirely for providers with no loaded connection. */ state: ProviderHealthState; + /** + * What the provider actually served over a window, from `call_logs`. Present + * only when the caller asks for it (`withUsage`). Complements `state`, which + * describes the connection right now and cannot see a provider that answers + * every call with an error. + */ + usage?: ProviderUsage; } +export interface ProviderUsage { + requests: number; + successes: number; + /** `null` below `MIN_USAGE_REQUESTS` — too small a sample to state a rate. */ + successRate: number | null; + avgLatencyMs: number | null; + lastRequestAt: string | null; + windowHours: number; +} + +/** + * Below this many requests in the window, no rate is reported: 1 failure out of + * 2 calls is not "50% broken", and a provider nobody called is not "0% healthy". + */ +const MIN_USAGE_REQUESTS = 5; + /** * Options controlling the additive "configured" / "available" filters. * Both default off (undefined/false) → output identical to current behavior. @@ -259,6 +287,14 @@ export interface FreeProviderRankingFilterOptions { configuredOnly?: boolean; /** Keep only providers that have ≥1 non-exhausted, non-rate-limited connection (implies configured). */ availableOnly?: boolean; + /** + * Also report what each provider actually served (`reliability.usage`). + * Off by default: it costs one aggregate query over `call_logs`, which a + * caller that only needs the ranking should not pay. + */ + withUsage?: boolean; + /** Window for `withUsage`. Defaults to `24h`, the health matrix's own default. */ + usageRange?: ProviderHealthMatrixRange; } /** Group connection states by provider id (shared by filter and reliability attach). */ @@ -381,6 +417,37 @@ export function attachProviderReliability( }); } +/** + * Pure enrichment: attach `usage` to the `reliability` of every ranking that has + * a row in the windowed aggregate. Rankings without `reliability` (no connection + * loaded) are returned unchanged, never mutated. + */ +export function attachProviderUsage( + rankings: FreeProviderRanking[], + usageRows: ProviderUsageRow[], + windowHours: number +): FreeProviderRanking[] { + const byProvider = new Map(usageRows.map((row) => [row.provider, row])); + return rankings.map((ranking) => { + const row = byProvider.get(ranking.id); + if (!row || !ranking.reliability) return ranking; + return { + ...ranking, + reliability: { + ...ranking.reliability, + usage: { + requests: row.requests, + successes: row.successes, + successRate: row.requests >= MIN_USAGE_REQUESTS ? row.successes / row.requests : null, + avgLatencyMs: row.avgLatencyMs ?? null, + lastRequestAt: row.lastRequestAt ?? null, + windowHours, + }, + }, + }; + }); +} + /** * Compute rankings for free providers based on ELO scores. * @@ -474,6 +541,20 @@ export async function computeFreeProviderRankings( // `availableOnly` already drops providers with no healthy connection, so under // it `state` is never `down`; `down` needs `configuredOnly` alone. filtered = attachProviderReliability(filtered, connections); + + // Third dimension, opt-in: what the provider actually served. `state` above + // reads the connection as it stands now and cannot see a provider that + // answers every call with an error — only the call log can. + if (opts.withUsage) { + const range = opts.usageRange ?? "24h"; + const windowMs = RANGE_MS[range]; + const since = new Date(Date.now() - windowMs).toISOString(); + filtered = attachProviderUsage( + filtered, + getProviderUsageSince(since), + windowMs / (60 * 60 * 1000) + ); + } } return filtered.slice(0, limit); diff --git a/src/lib/memory/injection.ts b/src/lib/memory/injection.ts index 9fdd2bbbf10..d4d8ead7f7c 100644 --- a/src/lib/memory/injection.ts +++ b/src/lib/memory/injection.ts @@ -65,7 +65,11 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * * Populated with the Xiaomi MiMo endpoint (provider id `xiaomi-mimo`, registry * alias `mimo`, serving mimo-v2.5) confirmed live to 400 on a non-first system - * message. Add other providers here only when they are documented as strict. + * message, and the TokenRouter gateway (provider id `tokenrouter`), confirmed + * live on 2026-08-22 to reject mid-array system messages — including the + * compression notice spliced by purifyHistory() before that splice was fixed to + * merge into the leading system message. Add other providers here only when + * they are documented as strict. * * Self-hosted deployments can extend this list without a source change via * OMNIROUTE_STRICT_SYSTEM_PROVIDERS (comma-separated provider ids, @@ -73,7 +77,7 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * self-hosted Qwen3.5+/3.6 model, whose chat template enforces the same * single-leading-system-message constraint as xiaomi-mimo. */ -const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo"]); +const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo", "tokenrouter"]); /** * Parses OMNIROUTE_STRICT_SYSTEM_PROVIDERS into a normalized id list. diff --git a/src/lib/modelCapabilities.ts b/src/lib/modelCapabilities.ts index 727122d2048..9d9572a7b13 100644 --- a/src/lib/modelCapabilities.ts +++ b/src/lib/modelCapabilities.ts @@ -254,6 +254,39 @@ function leafModelId(modelId: string | null | undefined): string | null { return leaf && leaf !== modelId ? leaf : null; } +/** + * Effort suffixes the catalog synthesizes as `-` variant ids from a + * base model's `supportedThinkingEfforts` (mirrors REGISTERED_EFFORT_SUFFIXES + * in open-sse/utils/registeredEffortVariants.ts, plus `minimal` for muse). + */ +const EFFORT_VARIANT_SUFFIXES = [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", +] as const; + +/** + * Strip a trailing effort-tier suffix off a model id (e.g. + * `deepseek-v4-flash-max` → `deepseek-v4-flash`). Longest token first so + * `xhigh` is matched before `high`. Returns null when no known suffix matches + * or the id would be left empty. + */ +function stripKnownEffortSuffix(modelId: string): string | null { + const normalized = String(modelId || "").trim(); + if (!normalized) return null; + for (const suffix of EFFORT_VARIANT_SUFFIXES) { + const token = `-${suffix}`; + if (normalized.length > token.length && normalized.endsWith(token)) { + return normalized.slice(0, -token.length); + } + } + return null; +} + function getStaticSpec(modelId: string | null, rawModel: string | null): ModelSpec | undefined { if (modelId) { const byCanonical = getModelSpec(modelId); @@ -704,15 +737,38 @@ export function getResolvedModelCapabilities( // persisted override never feeds back into the comparison that (re)writes it. const usePersistedOverrides = options?.persistedOverrides !== false; const resolved = resolveCapabilityInput(input); - const spec = getStaticSpec(resolved.model, resolved.rawModel); - const registryModel = getRegistryModel(resolved.provider, resolved.model); - const synced = getSyncedCapabilityForResolved( + let spec = getStaticSpec(resolved.model, resolved.rawModel); + let registryModel = getRegistryModel(resolved.provider, resolved.model); + let synced = getSyncedCapabilityForResolved( resolved.provider, resolved.model, resolved.rawModel, snapshot ); + // Effort-suffixed variants (e.g. command-code `deepseek-v4-flash-max`, + // `meta/muse-spark-1.2-contributor-xhigh`) are synthesized in the catalog + // from the base model's `supportedThinkingEfforts`; they have no registry + // row, synced row, or static spec of their own. Without a base-model + // fallback the variant resolves with NULL tool/vision/context capabilities, + // so a tool-bearing combo request treats the target as incompatible and + // silently reorders it behind models with confirmed capabilities. Resolve + // the variant's capabilities from its base model when every direct source + // misses. + if (!spec && !registryModel && !synced && resolved.provider && resolved.model) { + const baseModelId = stripKnownEffortSuffix(resolved.model); + if (baseModelId && baseModelId !== resolved.model) { + spec = getStaticSpec(baseModelId, resolved.rawModel); + registryModel = getRegistryModel(resolved.provider, baseModelId); + synced = getSyncedCapabilityForResolved( + resolved.provider, + baseModelId, + resolved.rawModel, + snapshot + ); + } + } + const modalitiesInput = parseModalities(synced?.modalities_input); const modalitiesOutput = parseModalities(synced?.modalities_output); const lookupKey = diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 6ed9785d73e..b7aa2aa2164 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -130,6 +130,11 @@ function uniqueStrings(values: Array) { ]; } +export function isGlmFamilyModel(modelId: string, displayName = ""): boolean { + const glmFamilyPattern = /(?:^|[/@:_. -])glm(?=$|[-._ /@:](?:z)?\d|\d)/i; + return glmFamilyPattern.test(modelId) || glmFamilyPattern.test(displayName); +} + function toQualifiedId( providerAlias: string | null, provider: string | null, @@ -457,6 +462,36 @@ export function enrichCatalogModelEntry( { provider, model }, capabilitySnapshot ); + const existingCapabilities = + entry.capabilities && typeof entry.capabilities === "object" + ? (entry.capabilities as JsonRecord) + : {}; + const declaredEffortTiers = Array.isArray(existingCapabilities.effort_tiers) + ? existingCapabilities.effort_tiers.filter( + (effort): effort is string => typeof effort === "string" && effort.length > 0 + ) + : []; + const sourceDeclaresThinking = + typeof existingCapabilities.thinking === "boolean" || + typeof existingCapabilities.supportsThinking === "boolean"; + const effortTiers = + metadata.capabilities.supportedThinkingEfforts && + metadata.capabilities.supportedThinkingEfforts.length > 0 + ? [...metadata.capabilities.supportedThinkingEfforts] + : declaredEffortTiers.length > 0 + ? declaredEffortTiers + : sourceDeclaresThinking + ? undefined + : // #10963: GLM-family models never inherit generic OpenAI tiers — an + // explicit empty list is authoritative unless a provider-declared + // contract exists (handled by declaredEffortTiers above). + isGlmFamilyModel(metadata.model, metadata.displayName) + ? [] + : extendCodexGpt56EffortValues( + metadata.provider, + metadata.model, + CANONICAL_EFFORT_VALUES + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -477,23 +512,15 @@ export function enrichCatalogModelEntry( // #6241: surface thinking support + the canonical effort tiers so the frontend can // render the effort/thinking toggles. `thinking` is kept for back-compat; `supportsThinking` // is the explicit flag and `effort_tiers` lists the selectable reasoning levels - // (only when the model actually supports thinking). + // (only when the model actually supports thinking). An explicit empty registry list + // is authoritative; GLM models also require a provider-declared contract instead of + // inheriting generic OpenAI effort tiers. ...(typeof metadata.capabilities.supportsThinking === "boolean" ? { thinking: metadata.capabilities.supportsThinking, supportsThinking: metadata.capabilities.supportsThinking, - ...(metadata.capabilities.supportsThinking - ? { - effort_tiers: - metadata.capabilities.supportedThinkingEfforts && - metadata.capabilities.supportedThinkingEfforts.length > 0 - ? [...metadata.capabilities.supportedThinkingEfforts] - : extendCodexGpt56EffortValues( - metadata.provider, - metadata.model, - CANONICAL_EFFORT_VALUES - ), - } + ...(metadata.capabilities.supportsThinking && effortTiers + ? { effort_tiers: effortTiers } : {}), } : {}), @@ -509,9 +536,7 @@ export function enrichCatalogModelEntry( }; nextEntry.capabilities = { - ...(entry.capabilities && typeof entry.capabilities === "object" - ? (entry.capabilities as JsonRecord) - : {}), + ...existingCapabilities, ...capabilityFields, }; diff --git a/src/lib/monitoring/comboHealthAutopilot.ts b/src/lib/monitoring/comboHealthAutopilot.ts index 71a10494837..24eba32a634 100644 --- a/src/lib/monitoring/comboHealthAutopilot.ts +++ b/src/lib/monitoring/comboHealthAutopilot.ts @@ -16,6 +16,7 @@ import type { ComboForecastMetrics, ComboForecastResponse, ComboForecastRiskLevel, + ProviderAutopilotReport, ComboHealthMetrics, ComboHealthResponse, ComboRecord, @@ -34,6 +35,7 @@ export interface ComboHealthAutopilotOptions { combos?: ComboRecord[]; healthResponse?: ComboHealthResponse; forecastResponse?: ComboForecastResponse; + providerHealthResponse?: ProviderAutopilotReport; } type ProviderIssueView = { @@ -103,7 +105,12 @@ function actionSet( case "open_combo_editor": return action(type, "Open combo editor", target, "/dashboard/combos"); case "run_combo_test": - return action(type, "Run combo test", target, "/dashboard/combos"); + return action( + type, + "Run combo test", + target, + `/dashboard/combos?test=${encodeURIComponent(target.comboId)}` + ); case "open_provider_health_autopilot": return action(type, "Open provider autopilot", target, "/dashboard/health"); case "review_quota_limits": @@ -447,7 +454,8 @@ export async function buildComboHealthAutopilotReport( now: options.now, combos: combosSnapshot, }), - buildProviderHealthAutopilotReport({ includeHealthy: false, includeActions: false }), + options.providerHealthResponse ?? + buildProviderHealthAutopilotReport({ includeHealthy: false, includeActions: false }), ]); const forecastsByComboId = new Map(forecast.combos.map((entry) => [entry.comboId, entry])); @@ -470,7 +478,7 @@ export async function buildComboHealthAutopilotReport( const degradedCount = allCombos.filter((combo) => combo.state === "degraded").length; const healthyCount = allCombos.filter((combo) => combo.state === "healthy").length; const issueCount = allCombos.reduce((sum, combo) => sum + combo.issues.length, 0); - const actionableCount = allCombos.reduce( + const suggestionCount = allCombos.reduce( (sum, combo) => sum + combo.issues.reduce((issueSum, issue) => issueSum + issue.actions.length, 0), 0 @@ -487,7 +495,8 @@ export async function buildComboHealthAutopilotReport( degradedCount, downCount, issueCount, - actionableCount, + suggestionCount, + actionableCount: suggestionCount, }, combos, }; diff --git a/src/lib/monitoring/providerHealthMatrix.ts b/src/lib/monitoring/providerHealthMatrix.ts index b1ff287baee..f9af936417a 100644 --- a/src/lib/monitoring/providerHealthMatrix.ts +++ b/src/lib/monitoring/providerHealthMatrix.ts @@ -121,7 +121,8 @@ interface CallLogTargetStats { lastErrorStatus: number | null; } -const RANGE_MS: Record = { +/** Exported so other surfaces reporting over a window use the same scale. */ +export const RANGE_MS: Record = { "1h": 60 * 60 * 1000, "24h": 24 * 60 * 60 * 1000, "7d": 7 * 24 * 60 * 60 * 1000, @@ -364,15 +365,15 @@ export async function buildProviderHealthMatrix( Promise.resolve(queryCallLogTargetStats(cutoff, null)), ]); - const connectionRows = (connections as JsonRecord[]).filter((connection) => { + const connectionRows = (connections as unknown as JsonRecord[]).filter((connection) => { const provider = canonicalProviderId(connection.provider); return provider && (!providerFilter || provider === providerFilter); }); - const breakerRows = (breakers as JsonRecord[]).filter((breaker) => { + const breakerRows = (breakers as unknown as JsonRecord[]).filter((breaker) => { const provider = canonicalProviderId(breaker.name); return provider && (!providerFilter || provider === providerFilter); }); - const lockoutRows = (lockouts as JsonRecord[]).filter((lockout) => { + const lockoutRows = (lockouts as unknown as JsonRecord[]).filter((lockout) => { const provider = canonicalProviderId(lockout.provider); return provider && (!providerFilter || provider === providerFilter); }); @@ -625,9 +626,9 @@ export async function buildProviderHealthMatrix( pools: poolReport.providers.map((p) => ({ provider: p.provider, health: p.health, - pool: p.pool as Record | null, - breaker: p.breaker as Record | null, - sessions: p.sessions as Array>, + pool: p.pool as unknown as Record | null, + breaker: p.breaker as unknown as Record | null, + sessions: p.sessions as unknown as Array>, issues: p.issues, })), }, diff --git a/src/lib/oauth/constants/oauth.ts b/src/lib/oauth/constants/oauth.ts index 0cace6c604c..0318db40b32 100644 --- a/src/lib/oauth/constants/oauth.ts +++ b/src/lib/oauth/constants/oauth.ts @@ -16,6 +16,7 @@ import { GROK_BUILD_TOKEN_URL, } from "@omniroute/open-sse/config/grokBuild.ts"; import { resolvePublicCred } from "@omniroute/open-sse/utils/publicCreds.ts"; +import { CURSOR_AGENT_CLI_VERSION } from "@omniroute/open-sse/utils/cursorAgentCliVersion.ts"; import { buildGitLabOAuthEndpoints, GITLAB_DUO_DEFAULT_BASE_URL } from "../gitlab"; /** @@ -364,20 +365,26 @@ export const KIRO_CONFIG = { authMethods: ["builder-id", "idc", "google", "github", "import"], }; -// Cursor OAuth Configuration (Import Token from Cursor IDE) +// Cursor OAuth Configuration (deep-control PKCE + optional IDE import) // Cursor stores credentials in SQLite database: state.vscdb -// Keys: cursorAuth/accessToken, storage.serviceMachineId +// Keys: cursorAuth/accessToken, cursorAuth/refreshToken, storage.serviceMachineId +// Deep-control PKCE + refresh aligned with OpenCodex (lidge-jun/opencodex src/oauth/cursor.ts). +// clientVersion pin lives in open-sse/utils/cursorAgentCliVersion.ts — single source of truth. export const CURSOR_CONFIG = { // API endpoints apiEndpoint: "https://api2.cursor.sh", chatEndpoint: "/aiserver.v1.ChatService/StreamUnifiedChatWithTools", - modelsEndpoint: "/aiserver.v1.AiService/GetDefaultModelNudgeData", + modelsEndpoint: "/aiserver.v1.AiService/AvailableModels", + // Standalone deep-control login (no IDE/CLI required) + loginUrl: "https://cursor.com/loginDeepControl", + pollUrl: "https://api2.cursor.sh/auth/poll", + refreshUrl: "https://api2.cursor.sh/auth/exchange_user_api_key", // Additional endpoints api3Endpoint: "https://api3.cursor.sh", // Telemetry agentEndpoint: "https://agent.api5.cursor.sh", // Privacy mode agentNonPrivacyEndpoint: "https://agentn.api5.cursor.sh", // Non-privacy mode - // Client metadata - clientVersion: "3.2.14", + // Client metadata — pin from cursorAgentCliVersion (not a second hardcoded string) + clientVersion: CURSOR_AGENT_CLI_VERSION, clientType: "ide", // Token storage locations (for user reference) tokenStoragePaths: { @@ -388,6 +395,7 @@ export const CURSOR_CONFIG = { // Database keys dbKeys: { accessToken: "cursorAuth/accessToken", + refreshToken: "cursorAuth/refreshToken", machineId: "storage.serviceMachineId", }, }; diff --git a/src/lib/oauth/providers/cursor.ts b/src/lib/oauth/providers/cursor.ts index 83ee0e5ae9a..7653acb5e24 100644 --- a/src/lib/oauth/providers/cursor.ts +++ b/src/lib/oauth/providers/cursor.ts @@ -3,13 +3,19 @@ import { CURSOR_CONFIG } from "../constants/oauth"; export const cursor = { config: CURSOR_CONFIG, flowType: "import_token", - mapTokens: (tokens) => ({ + mapTokens: (tokens: { + accessToken: string; + refreshToken?: string | null; + expiresIn?: number; + machineId?: string; + authMethod?: string; + }) => ({ accessToken: tokens.accessToken, - refreshToken: null, + refreshToken: tokens.refreshToken ?? null, expiresIn: tokens.expiresIn || 86400, providerSpecificData: { machineId: tokens.machineId, - authMethod: "imported", + authMethod: tokens.authMethod || (tokens.refreshToken ? "deep_control" : "imported"), }, }), }; diff --git a/src/lib/oauth/services/cursorLogin.ts b/src/lib/oauth/services/cursorLogin.ts new file mode 100644 index 00000000000..efc919a91a0 --- /dev/null +++ b/src/lib/oauth/services/cursorLogin.ts @@ -0,0 +1,245 @@ +/** + * Cursor deep-control PKCE login + refresh. + * Protocol aligned with OpenCodex `src/oauth/cursor.ts` (loginDeepControl + auth/poll + + * exchange_user_api_key). Verifiers stay server-side in the session store. + */ + +import { randomUUID } from "node:crypto"; +import { refreshCursorToken as refreshCursorTokenOpenSse } from "@omniroute/open-sse/services/tokenRefresh/providers/cursor.ts"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; +import { generatePKCE } from "../utils/pkce"; +import { CURSOR_CONFIG } from "../constants/oauth"; + +const SESSION_TTL_MS = 15 * 60 * 1000; +const EXPIRY_SKEW_MS = 5 * 60 * 1000; +const FALLBACK_TTL_MS = 60 * 60 * 1000; + +export type CursorAuthParams = { + verifier: string; + challenge: string; + uuid: string; + loginUrl: string; +}; + +export type CursorTokenCredentials = { + accessToken: string; + refreshToken: string; + expiresAt: Date; + accountId?: string; + email?: string; +}; + +type CursorJwtPayload = { + sub?: unknown; + email?: unknown; + exp?: unknown; +}; + +type StoredSession = { + verifier: string; + challenge: string; + uuid: string; + loginUrl: string; + createdAt: number; + expiresAt: number; +}; + +const sessions = new Map(); + +function decodeCursorJwtPayload(token: string): CursorJwtPayload | undefined { + const parts = token.split("."); + const payload = parts[1]; + if (parts.length !== 3 || !payload) return undefined; + try { + return JSON.parse(Buffer.from(payload, "base64url").toString("utf-8")) as CursorJwtPayload; + } catch { + return undefined; + } +} + +function cursorJwtIdentity(value: unknown): string | undefined { + if (typeof value === "string" && value.length > 0) return value; + if (typeof value === "number" && Number.isSafeInteger(value)) return String(value); + return undefined; +} + +function cursorJwtEmail(value: unknown): string | undefined { + if (typeof value !== "string" || value.length === 0) return undefined; + return value.toLowerCase(); +} + +/** Resolve token expiry (Date) from JWT `exp` minus skew; ~1h fallback. */ +export function getCursorTokenExpiry(token: string): Date { + const decoded = decodeCursorJwtPayload(token); + if (typeof decoded?.exp === "number") { + return new Date(decoded.exp * 1000 - EXPIRY_SKEW_MS); + } + return new Date(Date.now() + FALLBACK_TTL_MS); +} + +/** Build credentials from Cursor tokens, extracting stable identity from JWT `sub`. */ +export function credentialsFromCursorTokens( + accessToken: string, + refreshToken: string +): CursorTokenCredentials { + const payload = decodeCursorJwtPayload(accessToken) ?? decodeCursorJwtPayload(refreshToken); + const accountId = cursorJwtIdentity(payload?.sub); + const email = cursorJwtEmail(payload?.email); + return { + accessToken, + refreshToken, + expiresAt: getCursorTokenExpiry(accessToken), + ...(accountId ? { accountId } : {}), + ...(email ? { email } : {}), + }; +} + +/** Generate PKCE params + deep-control login URL (challenge only — never the verifier). */ +export async function generateCursorAuthParams(): Promise { + const { codeVerifier, codeChallenge } = generatePKCE(); + const uuid = randomUUID(); + const params = new URLSearchParams({ + challenge: codeChallenge, + uuid, + mode: "login", + redirectTarget: "cli", + }); + return { + verifier: codeVerifier, + challenge: codeChallenge, + uuid, + loginUrl: `${CURSOR_CONFIG.loginUrl}?${params.toString()}`, + }; +} + +function pruneExpiredSessions(now = Date.now()): void { + for (const [id, session] of sessions) { + if (session.expiresAt <= now) sessions.delete(id); + } +} + +export function clearCursorLoginSessions(): void { + sessions.clear(); +} + +/** + * Store verifier server-side. Returns public sessionId + loginUrl only. + * Multi-replica deployments need sticky sessions or a shared store. + */ +export function createCursorLoginSession(params: CursorAuthParams): { + sessionId: string; + loginUrl: string; +} { + pruneExpiredSessions(); + const sessionId = randomUUID(); + const now = Date.now(); + sessions.set(sessionId, { + verifier: params.verifier, + challenge: params.challenge, + uuid: params.uuid, + loginUrl: params.loginUrl, + createdAt: now, + expiresAt: now + SESSION_TTL_MS, + }); + return { sessionId, loginUrl: params.loginUrl }; +} + +/** Public view — never includes verifier. */ +export function getCursorLoginSession( + sessionId: string +): { uuid: string; loginUrl: string; expiresAt: number } | null { + pruneExpiredSessions(); + const session = sessions.get(sessionId); + if (!session) return null; + return { uuid: session.uuid, loginUrl: session.loginUrl, expiresAt: session.expiresAt }; +} + +/** Peek full session (for poll) without deleting. */ +export function peekCursorLoginSession(sessionId: string): StoredSession | null { + pruneExpiredSessions(); + return sessions.get(sessionId) ?? null; +} + +/** Consume (delete) full session — used after successful login or cancel. */ +export function consumeCursorLoginSession(sessionId: string): StoredSession | null { + pruneExpiredSessions(); + const session = sessions.get(sessionId) ?? null; + if (session) sessions.delete(sessionId); + return session; +} + +export function cancelCursorLoginSession(sessionId: string): boolean { + pruneExpiredSessions(); + return sessions.delete(sessionId); +} + +export type PollCursorOnceResult = + | { status: "pending" } + | { status: "ok"; accessToken: string; refreshToken: string } + | { status: "error"; message: string; httpStatus?: number }; + +/** + * Single poll attempt against Cursor auth/poll. + * 404 = still pending; 200 = tokens. Route layer owns the UI poll loop. + */ +export async function pollCursorAuthOnce( + uuid: string, + verifier: string, + signal?: AbortSignal +): Promise { + const url = `${CURSOR_CONFIG.pollUrl}?uuid=${encodeURIComponent(uuid)}&verifier=${encodeURIComponent(verifier)}`; + try { + const response = await fetch(url, { signal }); + if (response.status === 404) return { status: "pending" }; + if (response.ok) { + const data = (await response.json()) as { accessToken?: string; refreshToken?: string }; + if (!data.accessToken || !data.refreshToken) { + return { status: "error", message: "Cursor auth response missing tokens" }; + } + return { status: "ok", accessToken: data.accessToken, refreshToken: data.refreshToken }; + } + return { + status: "error", + message: `Cursor auth poll failed: ${response.status}`, + httpStatus: response.status, + }; + } catch (err) { + if (signal?.aborted) { + return { status: "error", message: "Cursor login cancelled" }; + } + const msg = err instanceof Error ? err.message : String(err); + return { status: "error", message: sanitizeErrorMessage(msg) }; + } +} + +export type RefreshCursorOptions = { + retryBaseMs?: number; + attempts?: number; +}; + +export type RefreshCursorResult = + CursorTokenCredentials | { error: "unrecoverable_refresh_error"; code: string } | null; + +/** + * Exchange a refresh token for fresh credentials (delegates to open-sse provider). + * Keeps the old refresh if the server omits one. 401/403 fail fast as unrecoverable. + */ +export async function refreshCursorAccessToken( + refreshToken: string, + log?: { error?: (...args: unknown[]) => void; info?: (...args: unknown[]) => void }, + options: RefreshCursorOptions = {} +): Promise { + const result = await refreshCursorTokenOpenSse(refreshToken, log, null, { + retryBaseMs: options.retryBaseMs, + attempts: options.attempts, + }); + if (!result) return null; + if ("error" in result) { + return { error: "unrecoverable_refresh_error", code: String(result.code || "unauthorized") }; + } + return { + accessToken: result.accessToken, + refreshToken: result.refreshToken, + expiresAt: new Date(result.expiresAt), + }; +} diff --git a/src/lib/oauth/services/persistCursorConnection.ts b/src/lib/oauth/services/persistCursorConnection.ts new file mode 100644 index 00000000000..1e577cd240c --- /dev/null +++ b/src/lib/oauth/services/persistCursorConnection.ts @@ -0,0 +1,76 @@ +/** + * Persist Cursor OAuth credentials (deep-control login or improved import). + */ + +import { + createProviderConnection, + getProviderConnections, + updateProviderConnection, +} from "@/models"; +import type { CursorTokenCredentials } from "./cursorLogin"; + +export type PersistCursorAuthMethod = "deep_control" | "imported" | "cursor-agent"; + +export type PersistCursorConnectionInput = CursorTokenCredentials & { + machineId?: string | null; + authMethod: PersistCursorAuthMethod; +}; + +function readAccountId(psd: unknown): string | null { + if (!psd || typeof psd !== "object") return null; + const rec = psd as Record; + for (const key of ["accountId", "userId"]) { + const v = rec[key]; + if (typeof v === "string" && v.length > 0) return v; + } + return null; +} + +/** + * Create or update a Cursor connection. Prefers accountId (JWT sub) match for + * multi-account safety; falls back to createProviderConnection email upsert. + */ +export async function persistCursorConnection(input: PersistCursorConnectionInput) { + const providerSpecificData = { + machineId: input.machineId || null, + authMethod: input.authMethod, + provider: input.authMethod === "deep_control" ? "Deep Control" : "Imported", + accountId: input.accountId || null, + userId: input.accountId || null, + ...(input.accountId ? { username: input.accountId } : {}), + }; + + if (input.accountId) { + const existing = (await getProviderConnections({ provider: "cursor" })) as Array<{ + id: string; + providerSpecificData?: unknown; + }>; + const match = existing.find( + (row) => readAccountId(row.providerSpecificData) === input.accountId + ); + if (match) { + return updateProviderConnection(match.id, { + accessToken: input.accessToken, + refreshToken: input.refreshToken, + expiresAt: input.expiresAt.toISOString(), + email: input.email || undefined, + providerSpecificData, + testStatus: "active", + lastError: null, + lastErrorType: null, + errorCode: null, + }); + } + } + + return createProviderConnection({ + provider: "cursor", + authType: "oauth", + accessToken: input.accessToken, + refreshToken: input.refreshToken, + expiresAt: input.expiresAt.toISOString(), + email: input.email || null, + providerSpecificData, + testStatus: "active", + }); +} diff --git a/src/lib/providerModels/cursorAvailableModels.ts b/src/lib/providerModels/cursorAvailableModels.ts new file mode 100644 index 00000000000..0fdc3d1939a --- /dev/null +++ b/src/lib/providerModels/cursorAvailableModels.ts @@ -0,0 +1,188 @@ +/** + * Fetch Cursor Available Models via HTTP (no cursor-agent binary required). + * Uses Connect-RPC JSON against api2.cursor.sh AiService/AvailableModels. + */ + +import { CURSOR_CONFIG } from "@/lib/oauth/constants/oauth"; +import { CursorService } from "@/lib/oauth/services/cursor"; +import { + humanizeCursorModelId, + type CursorAgentModelEntry, +} from "@/lib/providerModels/cursorAgent"; +import { getConsistentMachineId } from "@/shared/utils/machineId"; + +export type FetchCursorAvailableModelsOptions = { + accessToken: string; + machineId?: string | null; + fetchImpl?: typeof fetch; + signal?: AbortSignal; +}; + +function asRecord(value: unknown): Record | null { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : null; +} + +function pickModelId(entry: Record): string | null { + for (const key of ["name", "modelId", "model_id", "id", "slug"]) { + const v = entry[key]; + if (typeof v === "string" && v.trim()) return v.trim(); + } + return null; +} + +function pickModelName(entry: Record, id: string): string { + for (const key of ["displayName", "display_name", "title", "label"]) { + const v = entry[key]; + if (typeof v === "string" && v.trim()) return v.trim(); + } + return humanizeCursorModelId(id); +} + +/** + * Normalize AvailableModels JSON (Connect JSON or protobuf-json) into catalog rows. + * Exported for unit tests. + * + * Always ensures catalog id `auto` is present (Cursor often returns wire id `default` + * only). OmniRoute clients request `cu/auto`; resolveRequestedModel maps it to `default`. + */ +export function normalizeCursorAvailableModelsPayload(payload: unknown): CursorAgentModelEntry[] { + const root = asRecord(payload) ?? {}; + const candidates: unknown[] = []; + + for (const key of ["models", "availableModels", "available_models", "model"]) { + const v = root[key]; + if (Array.isArray(v)) candidates.push(...v); + } + + // Some Connect JSON responses nest under `models.models` or similar + const nestedModels = asRecord(root.models); + if (nestedModels) { + for (const key of ["models", "items", "list"]) { + const v = nestedModels[key]; + if (Array.isArray(v)) candidates.push(...v); + } + } + + if (Array.isArray(payload)) candidates.push(...payload); + + const seen = new Set(); + const out: CursorAgentModelEntry[] = []; + for (const item of candidates) { + if (typeof item === "string" && item.trim()) { + const id = item.trim(); + if (seen.has(id)) continue; + seen.add(id); + out.push({ id, name: humanizeCursorModelId(id), owned_by: "cursor" }); + continue; + } + const rec = asRecord(item); + if (!rec) continue; + const id = pickModelId(rec); + if (!id || seen.has(id)) continue; + // Prefer usable / non-disabled when flags exist + if (rec.disabled === true || rec.isDisabled === true) continue; + if (rec.usable === false || rec.isUsable === false) continue; + seen.add(id); + out.push({ id, name: pickModelName(rec, id), owned_by: "cursor" }); + } + + return ensureCursorAutoCatalogEntry(out); +} + +/** OpenCodex-style Cursor Router optimization modes (catalog ids). */ +export const CURSOR_AUTO_ROUTER_VARIANT_IDS = [ + "auto-cost", + "auto-balance", + "auto-intelligence", +] as const; + +const CURSOR_AUTO_ROUTER_VARIANT_NAMES: Record< + (typeof CURSOR_AUTO_ROUTER_VARIANT_IDS)[number], + string +> = { + "auto-cost": "Auto (cost)", + "auto-balance": "Auto (balance)", + "auto-intelligence": "Auto (intelligence)", +}; + +/** Cursor auto-router: catalog id `auto`, wire id `default`. Always keep `auto` visible. */ +export function ensureCursorAutoCatalogEntry( + models: CursorAgentModelEntry[] +): CursorAgentModelEntry[] { + const byId = new Map(models.map((m) => [m.id, m])); + const out = [...models]; + + if (!byId.has("auto")) { + const defaultEntry = byId.get("default"); + const autoEntry: CursorAgentModelEntry = { + id: "auto", + name: defaultEntry?.name || "Auto (current, default)", + owned_by: "cursor", + }; + // Prefer `auto` as the public id; keep `default` for wire-compat listings. + out.unshift(autoEntry); + byId.set("auto", autoEntry); + } + + // Always expose Cost/Balance/Intelligence router modes (OpenCodex CURSOR_ROUTER_MODEL_IDS). + for (const id of CURSOR_AUTO_ROUTER_VARIANT_IDS) { + if (byId.has(id)) continue; + const entry: CursorAgentModelEntry = { + id, + name: CURSOR_AUTO_ROUTER_VARIANT_NAMES[id], + owned_by: "cursor", + }; + out.push(entry); + byId.set(id, entry); + } + + return out; +} + +export async function fetchCursorAvailableModels( + options: FetchCursorAvailableModelsOptions +): Promise { + const { accessToken, signal } = options; + if (!accessToken) throw new Error("Cursor access token is required for AvailableModels"); + + const machineId = options.machineId || (await getConsistentMachineId()); + const cursorService = new CursorService(); + const headers = { + ...cursorService.buildHeaders(accessToken, machineId), + // Prefer Connect JSON so we can parse without a protobuf schema + "Content-Type": "application/json", + Accept: "application/json", + }; + + const url = `${CURSOR_CONFIG.apiEndpoint}${CURSOR_CONFIG.modelsEndpoint}`; + const fetchImpl = options.fetchImpl || fetch; + const response = await fetchImpl(url, { + method: "POST", + headers, + body: "{}", + signal, + }); + + if (!response.ok) { + const text = await response.text().catch(() => ""); + throw new Error( + `Cursor AvailableModels failed: ${response.status}${text ? ` ${text.slice(0, 200)}` : ""}` + ); + } + + const contentType = response.headers.get("content-type") || ""; + if (contentType.includes("proto") || contentType.includes("protobuf")) { + throw new Error( + "Cursor AvailableModels returned protobuf; JSON catalog unavailable for this client version" + ); + } + + const payload = await response.json(); + const models = normalizeCursorAvailableModelsPayload(payload); + if (models.length === 0) { + throw new Error("Cursor AvailableModels returned no models"); + } + return models; +} diff --git a/src/lib/providerModels/modelDiscovery.ts b/src/lib/providerModels/modelDiscovery.ts index 923e3b8509a..85e605daff4 100644 --- a/src/lib/providerModels/modelDiscovery.ts +++ b/src/lib/providerModels/modelDiscovery.ts @@ -68,6 +68,19 @@ export function detectVisionInput(record: JsonRecord): boolean { // import format already emits). Hard Rule #7 — validate the untrusted upstream // payload with Zod before it is trusted/stored; a malformed shape degrades to // `undefined` instead of throwing, so one bad record never fails the whole sync. +// The same nesting also carries `default_effort` (e.g. OpenRouter +// `reasoning:{mandatory, default_enabled, default_effort, supported_efforts}`) — +// captured by `detectDefaultThinkingEffort` below and threaded through the +// EXISTING `defaultThinkingEffort` plumbing (`SyncedAvailableModel`, +// RuntimeModelMeta, #6879 `applyDefaultReasoningEffort`), so a model that only +// produces usable output with an explicit effort (measured: OpenRouter stealth +// reasoning models returning `upstream_empty_response` without one) gets the +// vendor-declared default injected instead of failing. +const reasoningDefaultEffortSchema = z + .object({ default_effort: z.string().optional() }) + .partial() + .nullable() + .optional(); const reasoningSupportedEffortsSchema = z .object({ supported_efforts: z.array(z.string()).optional() }) .partial() @@ -135,6 +148,28 @@ function parseEffortList(rawList: unknown): string[] | undefined { return efforts.length > 0 ? efforts : undefined; } +/** + * Read the nested `record.reasoning.default_effort` shape (OpenRouter declares + * `reasoning:{mandatory, default_enabled, default_effort, supported_efforts}`) + * and normalize it onto the canonical vocabulary (`max` → `xhigh`, same mapping + * `detectSupportedThinkingEfforts` applies to the tier list). Returns `undefined` + * (never throws) when the field is absent or malformed. + * + * A flat top-level `defaultThinkingEffort` (OmniRoute's own import format, and + * kimi-style upstreams) stays authoritative — the nested shape is a fallback. + */ +export function detectDefaultThinkingEffort(record: JsonRecord): string | undefined { + if (typeof record.defaultThinkingEffort === "string" && record.defaultThinkingEffort.length > 0) { + return normalizeSupportedEffort(record.defaultThinkingEffort); + } + const parsed = reasoningDefaultEffortSchema.safeParse(record.reasoning); + if (parsed.success && parsed.data) { + const raw = parsed.data.default_effort; + if (typeof raw === "string" && raw.length > 0) return normalizeSupportedEffort(raw); + } + return undefined; +} + /** * #7694: read the nested `record.reasoning.supported_efforts` shape and normalize each * tier onto the canonical vocabulary. Returns `undefined` (never throws) when the field @@ -251,6 +286,9 @@ export function normalizeDiscoveredModels( if (isCrofReasoningModel) return [...CROF_REASONING_EFFORTS]; return isCommandCodeModel ? [...COMMAND_CODE_REASONING_EFFORTS] : undefined; })(); + // Vendor-declared default effort (OpenRouter `reasoning.default_effort`, or the + // flat import field). Normalized onto the canonical vocabulary (`max` → `xhigh`). + const defaultThinkingEffort = detectDefaultThinkingEffort(record); const name = toNonEmptyString(record.name) || @@ -307,9 +345,7 @@ export function normalizeDiscoveredModels( : {}), ...(supportedEndpoints && supportedEndpoints.length > 0 ? { supportedEndpoints } : {}), ...(supportedThinkingEfforts !== undefined ? { supportedThinkingEfforts } : {}), - ...(toNonEmptyString(record.defaultThinkingEffort) - ? { defaultThinkingEffort: toNonEmptyString(record.defaultThinkingEffort)! } - : {}), + ...(defaultThinkingEffort !== undefined ? { defaultThinkingEffort } : {}), ...(typeof inputTokenLimit === "number" ? { inputTokenLimit } : {}), ...(typeof outputTokenLimit === "number" ? { outputTokenLimit } : {}), ...(typeof record.description === "string" ? { description: record.description } : {}), diff --git a/src/lib/providers/staticModels.ts b/src/lib/providers/staticModels.ts index bedda44f7b8..a62c90b1109 100644 --- a/src/lib/providers/staticModels.ts +++ b/src/lib/providers/staticModels.ts @@ -99,6 +99,15 @@ const STATIC_MODEL_PROVIDERS: Record Array<{ id: string; name: str { id: "google_scholar", name: "Google Scholar" }, { id: "duckduckgo", name: "DuckDuckGo" }, ], + "v0-vercel-web": () => [ + // v0-vercel-web web-cookie codegen provider — no upstream /v1/models endpoint, + // no registry `models` and no discovery config, so seed the current v0 lineup + // as a static catalog mirroring the v0-vercel API provider (shared.ts) so the + // model-import UI serves a list instead of the tail 400 (#10990). + { id: "v0-1.0-md", name: "V0 1.0 MD" }, + { id: "v0-1.5-lg", name: "V0 1.5 LG" }, + { id: "v0-1.5-md", name: "V0 1.5 MD" }, + ], "venice-web": () => [ // Venice.ai web-cookie provider — no upstream /v1/models endpoint, so seed the // current lineup as a static catalog (#6269). Venice rotates its catalog; keep @@ -114,6 +123,7 @@ const STATIC_MODEL_PROVIDERS: Record Array<{ id: string; name: str const SEARCH_TYPE_LABELS: Record = { web: "Web Search", news: "News Search", + x: "X Search", }; function formatSearchTypeLabel(searchType: string): string { diff --git a/src/lib/providers/validation.ts b/src/lib/providers/validation.ts index 7f8180dee00..8148173d1a0 100644 --- a/src/lib/providers/validation.ts +++ b/src/lib/providers/validation.ts @@ -108,6 +108,7 @@ import { validateBytezProvider, } from "./validation/webCookie"; import { validateAiHordeProvider } from "./validation/aihorde"; +import { validateDifyProvider } from "./validation/dify"; import { validateAdobeFireflyProvider } from "./validation/adobeFirefly"; import { validateV0VercelProvider, @@ -230,6 +231,10 @@ export async function validateProviderApiKey({ provider, apiKey, providerSpecifi freebuff: validateFreebuffProvider, "command-code": validateCommandCodeProvider, huggingface: validateHuggingFaceProvider, + // #11002: Dify serves no OpenAI-compatible route — only POST /v1/chat-messages. + // The generic OpenAI-like probe 404s on /v1/models and /v1/chat/completions, + // so every real app key was misreported as "endpoint not supported". + dify: validateDifyProvider, // #5422: auth-only probe — Bytez 404s on every chat model until the account adds it to // its catalog, so the generic chat probe can't validate a fresh key. bytez: validateBytezProvider, diff --git a/src/lib/providers/validation/dify.ts b/src/lib/providers/validation/dify.ts new file mode 100644 index 00000000000..3834af0e93c --- /dev/null +++ b/src/lib/providers/validation/dify.ts @@ -0,0 +1,86 @@ +/** + * Dify key check. Dify (multi-app LLM "LLMOps" platform) does NOT expose an + * OpenAI-compatible HTTP API. Its native completion endpoint is + * `POST {base}/v1/chat-messages` (body `inputs`/`query`/`response_mode`/`user` + * — no `model`/`messages` envelope). There is no `/v1/models` listing, so the + * generic OpenAI-like probe (GET /v1/models → POST /v1/chat/completions) + * always 404s and every real Dify app key is misreported as + * "Provider validation endpoint not supported" (#11002). + * + * Dify itself returns a clean 401 {"code":"unauthorized"} for a bad app key on + * `/v1/chat-messages`, and 200 for a valid key, so a single POST there is the + * correct auth probe. + */ +import { getRegistryEntry } from "@omniroute/open-sse/config/providerRegistry.ts"; +import { normalizeBaseUrl } from "./urlHelpers"; +import { toValidationErrorResult, validationWrite } from "./transport"; + +/** + * Shape a provider/connection base URL into the Dify native completion route. + * Accepts the cloud root (`https://api.dify.ai`), a `/v1` root, or a full + * `/v1/chat-messages` URL (e.g. a self-hosted instance) and always returns + * `{base}/v1/chat-messages`. + */ +export function resolveDifyChatMessagesUrl(baseUrl: string) { + const normalized = normalizeBaseUrl(baseUrl); + if (!normalized) return ""; + const cleaned = normalized.replace(/\/chat-messages$/, "").replace(/\/v1$/, ""); + return `${cleaned}/v1/chat-messages`; +} + +/** Pure status→verdict mapping, unit-testable without network. */ +export function difyValidationResultFromStatus(status: number) { + if (status === 401 || status === 403) { + return { valid: false, error: "Invalid API key" }; + } + if (status >= 200 && status < 300) { + return { valid: true, error: null }; + } + return { valid: false, error: `Dify validation failed (${status})` }; +} + +export async function validateDifyProvider({ + apiKey, + providerSpecificData = {}, + fetchImpl = validationWrite, +}: { + apiKey?: unknown; + providerSpecificData?: Record; + fetchImpl?: typeof validationWrite; +}) { + const key = typeof apiKey === "string" ? apiKey.trim() : ""; + if (!key) { + return { valid: false, error: "API key required" }; + } + + const specificBase = + typeof providerSpecificData?.baseUrl === "string" ? providerSpecificData.baseUrl.trim() : ""; + const entryBase = (getRegistryEntry("dify")?.baseUrl as string) || ""; + const probeUrl = resolveDifyChatMessagesUrl(specificBase || entryBase); + if (!probeUrl) { + return { valid: false, error: "Dify requires a Base URL" }; + } + + try { + const response = await fetchImpl( + probeUrl, + { + method: "POST", + headers: { + Authorization: `Bearer ${key}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + inputs: {}, + query: "ping", + response_mode: "blocking", + user: "omniroute-key-check", + }), + }, + false + ); + return difyValidationResultFromStatus(response.status); + } catch (error) { + return toValidationErrorResult(error); + } +} \ No newline at end of file diff --git a/src/lib/providers/validation/openaiFormat.ts b/src/lib/providers/validation/openaiFormat.ts index 3d601131345..fa7f2455797 100644 --- a/src/lib/providers/validation/openaiFormat.ts +++ b/src/lib/providers/validation/openaiFormat.ts @@ -1,7 +1,6 @@ // OpenAI/Gemini-format + Bedrock provider key validators (bedrock, openai-like, command-code, gemini-like, openai-compatible). // Extracted from validation.ts (god-file decomposition) — top-level functions; behavior is // byte-identical to the original inline defs. -import { randomUUID } from "node:crypto"; import { getRegistryEntry } from "@omniroute/open-sse/config/providerRegistry.ts"; import { discoverBedrockNativeModels, @@ -196,13 +195,12 @@ export async function validateOpenAILikeProvider({ export async function validateCommandCodeProvider({ apiKey, providerSpecificData = {} }: any) { const entry = getRegistryEntry("command-code"); const baseUrl = normalizeBaseUrl(entry?.baseUrl || "https://api.commandcode.ai"); - const chatPath = entry?.chatPath || "/alpha/generate"; + const chatPath = entry?.chatPath || "/provider/v1/chat/completions"; const url = `${baseUrl}${chatPath.startsWith("/") ? chatPath : `/${chatPath}`}`; const validationModelId = providerSpecificData?.validationModelId || entry?.models?.find((model) => model.id === "deepseek/deepseek-v4-flash")?.id || "deepseek/deepseek-v4-flash"; - const { COMMAND_CODE_VERSION } = await import("@omniroute/open-sse/executors/commandCode.ts"); return validateDirectChatProvider({ url, @@ -210,37 +208,13 @@ export async function validateCommandCodeProvider({ apiKey, providerSpecificData headers: { "Content-Type": "application/json", Authorization: `Bearer ${apiKey}`, - "x-command-code-version": COMMAND_CODE_VERSION, - "x-cli-environment": "external", - "x-project-slug": "pi-cc", - "x-taste-learning": "false", - "x-co-flag": "false", - "x-session-id": randomUUID(), + Accept: "text/event-stream", }, body: { - config: { - workingDir: "/workspace", - date: new Date().toISOString().slice(0, 10), - environment: "external", - structure: [], - isGitRepo: false, - currentBranch: "", - mainBranch: "", - gitStatus: "", - recentCommits: [], - }, - memory: "", - taste: "", - skills: "", - permissionMode: "standard", - params: { - model: validationModelId, - messages: [{ role: "user", content: "test" }], - tools: [], - system: "", - max_tokens: 1, - stream: true, - }, + model: validationModelId, + messages: [{ role: "user", content: "test" }], + stream: true, + max_tokens: 1, }, }); } diff --git a/src/lib/providers/validation/searchProviders.ts b/src/lib/providers/validation/searchProviders.ts index 302a7922662..67b8b36ae03 100644 --- a/src/lib/providers/validation/searchProviders.ts +++ b/src/lib/providers/validation/searchProviders.ts @@ -147,6 +147,19 @@ export const SEARCH_VALIDATOR_CONFIGS: Record< body: JSON.stringify({ query: "test", max_results: 1 }), }, }), + "x-search": (apiKey) => ({ + url: "https://api.x.ai/v1/responses", + init: { + method: "POST", + headers: { "Content-Type": "application/json", Authorization: `Bearer ${apiKey}` }, + body: JSON.stringify({ + model: "grok-4.6", + stream: false, + input: "test", + tools: [{ type: "x_search" }], + }), + }, + }), "zai-search": (apiKey, providerSpecificData = {}) => { const baseUrl = typeof providerSpecificData?.baseUrl === "string" && providerSpecificData.baseUrl.trim() diff --git a/src/lib/providers/validation/webProvidersA.ts b/src/lib/providers/validation/webProvidersA.ts index a1b20f1df9b..52e9aa4c0ef 100644 --- a/src/lib/providers/validation/webProvidersA.ts +++ b/src/lib/providers/validation/webProvidersA.ts @@ -13,11 +13,14 @@ import { normalizeSessionCookieHeader, } from "@/lib/providers/webCookieAuth"; -// kimi-web uses the international `www.kimi.com` Connect-RPC API. The legacy -// `kimi.moonshot.cn` domain now 307-redirects every non-CN visitor, and even -// if you bypass the redirect the old `/api/chat` REST endpoint is gone. The -// SPA exposes a profile probe at `GET /api/user` that returns the user object -// at the top level when the `Authorization: Bearer ` header is valid. +// kimi-web uses the international (west-facing) `www.kimi.ai` Connect-RPC API by +// default. `www.kimi.com` is the China-region endpoint — it serves China users but +// the China region is not reliably reachable from outside CN, so it is not the +// default. The legacy `kimi.moonshot.cn` domain now 307-redirects every non-CN +// visitor, and even if you bypass the redirect the old `/api/chat` REST endpoint is +// gone. The SPA exposes a profile probe at `GET /api/user` that returns the user +// object at the top level when the `Authorization: Bearer ` header is +// valid. Override the endpoint with KIMI_WEB_BASE_URL (opt-in). export async function validateKimiWebProvider({ apiKey }: any) { const rawCred = String(apiKey ?? "").trim(); if (!rawCred) { diff --git a/src/lib/quota/providerCapabilities.ts b/src/lib/quota/providerCapabilities.ts deleted file mode 100644 index cc5ec32a767..00000000000 --- a/src/lib/quota/providerCapabilities.ts +++ /dev/null @@ -1,37 +0,0 @@ -export interface ProviderCapabilities { - providerId: string; - quotaApi: boolean; - usageApi: boolean; - rateLimitHeaders: boolean; - streaming: boolean; - toolUse: boolean; - coding: boolean; - vision: boolean; - longContext: boolean; -} - -const registry = new Map(); - -export function registerProviderCapabilities(capabilities: ProviderCapabilities): void { - registry.set(capabilities.providerId, { ...capabilities }); -} - -export function getProviderCapabilities(providerId: string): ProviderCapabilities { - return ( - registry.get(providerId) ?? { - providerId, - quotaApi: false, - usageApi: false, - rateLimitHeaders: false, - streaming: false, - toolUse: false, - coding: false, - vision: false, - longContext: false, - } - ); -} - -export function listProviderCapabilities(): ProviderCapabilities[] { - return [...registry.values()].map((capabilities) => ({ ...capabilities })); -} diff --git a/src/lib/quota/providerQuotaTelemetry.ts b/src/lib/quota/providerQuotaTelemetry.ts index cd9a29f2ad1..05955690c6f 100644 --- a/src/lib/quota/providerQuotaTelemetry.ts +++ b/src/lib/quota/providerQuotaTelemetry.ts @@ -48,12 +48,6 @@ export interface ProviderConnectionForQuota { [key: string]: unknown; } -export interface ProviderQuotaMonitor { - providerId: string; - supportedDimensions(): Promise; - fetchQuotaState(connection: ProviderConnectionForQuota): Promise; -} - export type QuotaSourceKind = "provider_api" | "response_headers" | "configured" | "estimated" | "unknown"; diff --git a/src/lib/quota/redisQuotaStore.ts b/src/lib/quota/redisQuotaStore.ts index 7c0e99ceb7a..d9d17309eda 100644 --- a/src/lib/quota/redisQuotaStore.ts +++ b/src/lib/quota/redisQuotaStore.ts @@ -72,7 +72,7 @@ export function resetRedisClient(): void { // Key helpers // --------------------------------------------------------------------------- -const KEY_PREFIX = "omniroute:quota"; +const KEY_PREFIX = `${process.env.REDIS_KEY_PREFIX?.trim() || "omniroute:"}quota`; function bucketKey(apiKeyId: string, dimensionKey: string, bucketIndex: number): string { return `${KEY_PREFIX}:${apiKeyId}:${dimensionKey}:${bucketIndex}`; diff --git a/src/lib/search/executeWebSearch.ts b/src/lib/search/executeWebSearch.ts index 64565518d56..6e2af547f0d 100644 --- a/src/lib/search/executeWebSearch.ts +++ b/src/lib/search/executeWebSearch.ts @@ -7,7 +7,7 @@ import { resolveSearchProvider, selectProvider, supportsSearchType, - SEARCH_CREDENTIAL_FALLBACKS, + getSearchCredentialFallbacks, SEARCH_PROVIDERS, type SearchProviderConfig, } from "@omniroute/open-sse/config/searchRegistry.ts"; @@ -25,7 +25,7 @@ export interface ExecuteWebSearchInput { provider?: string; max_results?: number; limit?: number; - search_type?: "web" | "news"; + search_type?: "web" | "news" | "x"; offset?: number; country?: string; language?: string; @@ -64,8 +64,10 @@ export class WebSearchExecutionError extends Error { async function resolveSearchCredentials(providerId: string) { const creds = await getProviderCredentials(providerId).catch(() => null); if (creds) return creds; - const fallbackId = SEARCH_CREDENTIAL_FALLBACKS[providerId]; - if (fallbackId) return getProviderCredentials(fallbackId).catch(() => null); + for (const fallbackId of getSearchCredentialFallbacks(providerId)) { + const fallback = await getProviderCredentials(fallbackId).catch(() => null); + if (fallback) return fallback; + } return null; } @@ -108,7 +110,12 @@ function assertValidSearchInput(input: ExecuteWebSearchInput) { if (input.query.trim().length > 500) { throw new WebSearchExecutionError("Query must be 500 characters or fewer", 400); } - if (input.search_type && input.search_type !== "web" && input.search_type !== "news") { + if ( + input.search_type && + input.search_type !== "web" && + input.search_type !== "news" && + input.search_type !== "x" + ) { throw new WebSearchExecutionError(`Unsupported search_type: ${String(input.search_type)}`, 400); } } @@ -119,6 +126,8 @@ export async function executeWebSearch( assertValidSearchInput(input); const log = input.log || defaultLog; + if (input.provider === "x_search") input.provider = "x-search"; + if (input.provider === "x-search") input.search_type = "x"; const searchType = input.search_type || "web"; if (input.provider) { diff --git a/src/lib/services/ServiceSupervisor.ts b/src/lib/services/ServiceSupervisor.ts index 9ec8e44f6ab..df1011bb456 100644 --- a/src/lib/services/ServiceSupervisor.ts +++ b/src/lib/services/ServiceSupervisor.ts @@ -7,7 +7,12 @@ import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error"; import { getServiceRow, updateServiceField, setToolStatus } from "@/lib/db/versionManager"; import { RingBuffer } from "./ringBuffer"; import { HealthChecker } from "./healthCheck"; -import { decidePreSpawn, probeBeforeSpawn, resolvePortPid } from "./portProbe"; +import { + decidePreSpawn, + isAdoptExistingEnabled, + probeBeforeSpawn, + resolvePortPid, +} from "./portProbe"; import type { ServiceConfig, ServiceState, ServiceStatus, LogLine, HealthState } from "./types"; const CRASH_FAST_THRESHOLD_MS = 5_000; @@ -111,7 +116,7 @@ export class ServiceSupervisor extends EventEmitter { // Opt-in per ServiceConfig so the default spawn path is unchanged. if (this.config.probeBeforeSpawn) { const probe = await probeBeforeSpawn(this.config.healthUrl(), this.config.port); - const decision = decidePreSpawn(probe, this.config.port); + const decision = decidePreSpawn(probe, this.config.port, isAdoptExistingEnabled()); if (decision.action === "adopt") { // Something healthy already serves this port. We didn't spawn it, diff --git a/src/lib/services/installers/utils.ts b/src/lib/services/installers/utils.ts index aff9c8c6226..45bdef032f9 100644 --- a/src/lib/services/installers/utils.ts +++ b/src/lib/services/installers/utils.ts @@ -145,13 +145,16 @@ export function runNpm( options: { cwd?: string; timeoutMs?: number; prefix?: string } = {} ): Promise { const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS; - // On Windows, npm is npm.cmd; on Unix it's npm. - const npmBin = process.platform === "win32" ? "npm.cmd" : "npm"; + const isBun = Boolean(process.versions.bun); + const npmBin = process.platform === "win32" + ? (isBun ? "bun.exe" : "npm.cmd") + : (isBun ? "bun" : "npm"); + const execArgs = isBun && args[0] === "install" ? ["add", ...args.slice(1)] : args; return new Promise((resolve, reject) => { execFile( npmBin, - args, + execArgs, buildNpmExecOptions(process.platform, { cwd: options.cwd, timeoutMs, diff --git a/src/lib/services/portProbe.ts b/src/lib/services/portProbe.ts index 81c3d8a8467..2a9fd502bda 100644 --- a/src/lib/services/portProbe.ts +++ b/src/lib/services/portProbe.ts @@ -37,11 +37,30 @@ const PID_RESOLVE_TIMEOUT_MS = 2_000; * * Pure — no I/O — so it can be exhaustively unit-tested. */ -export function decidePreSpawn(probe: PreSpawnProbe, port: number): PreSpawnDecision { - // A healthy instance is already serving on the port — adopt it rather than - // spawn a duplicate that would immediately die with EADDRINUSE. +export function decidePreSpawn( + probe: PreSpawnProbe, + port: number, + allowAdopt = false +): PreSpawnDecision { if (probe.healthy) { - return { action: "adopt" }; + // A 2xx on the health path does NOT prove the listener is our service: a + // local process can squat the port, answer 200, and get adopted — receiving + // the injected service API key and script execution inside the dashboard + // origin (GHSA-wg9p-6m2g-4v27). Adopt an already-healthy listener only when + // the operator explicitly opts in; otherwise surface the same actionable + // error we already use for a held-but-unhealthy port instead of silently + // trusting the listener. + if (allowAdopt) { + return { action: "adopt" }; + } + return { + action: "error", + message: + `Port ${port} is already serving a healthy response, but adopting an ` + + `existing listener is disabled by default (a 2xx cannot prove the listener ` + + `is this service). Set OMNIROUTE_ADOPT_EXISTING_SERVICE=1 to allow adoption, ` + + `or stop the process holding the port and start the service again.`, + }; } // Port is held but nothing healthy answers: an orphaned or unrelated process // is squatting on it. Surface a clear, actionable error instead of letting @@ -59,6 +78,17 @@ export function decidePreSpawn(probe: PreSpawnProbe, port: number): PreSpawnDeci return { action: "spawn" }; } +/** + * Whether the operator opted in to adopting an already-healthy listener on a + * service port. Off by default (GHSA-wg9p-6m2g-4v27): a squatter can answer a + * 2xx, so auto-adoption is only safe when the operator knows the listener is + * genuinely their (externally-managed) instance. + */ +export function isAdoptExistingEnabled(env: NodeJS.ProcessEnv = process.env): boolean { + const v = env.OMNIROUTE_ADOPT_EXISTING_SERVICE; + return v === "1" || v === "true"; +} + /** TCP connect check: resolves true when something accepts a connection. */ function isPortInUse(port: number, timeoutMs: number): Promise { return new Promise((resolve) => { diff --git a/src/lib/tokenHealthCheck.ts b/src/lib/tokenHealthCheck.ts index 3e38eb3dbc3..4927c4823fc 100644 --- a/src/lib/tokenHealthCheck.ts +++ b/src/lib/tokenHealthCheck.ts @@ -718,10 +718,12 @@ export async function checkConnection(conn) { // cosmetic "Token Expired". Surface reality as a terminal "expired" status instead. // Guard tightly so we do NOT clobber: // - providers without refresh tokens (supportsTokenRefresh=false; #8407 devin-cli) + // - Cursor access-token-only imports (refresh is optional; deep-control stores one) // - connections already in a terminal/specific state (expired/banned/credits_exhausted) // - transient cooldown state (unavailable) owned by the request path const refreshCapableNeedsReauth = supportsTokenRefresh(conn.provider) && + conn.provider !== "cursor" && (!conn.testStatus || conn.testStatus === "active") && !(conn.apiKey && conn.apiKey.length > 0); // API-key-only connections don't need refresh tokens if (refreshCapableNeedsReauth) { diff --git a/src/lib/usage/callLogRotation.ts b/src/lib/usage/callLogRotation.ts index a5d63a5d7e3..58ac3066ad1 100644 --- a/src/lib/usage/callLogRotation.ts +++ b/src/lib/usage/callLogRotation.ts @@ -20,6 +20,7 @@ import { type CallLogDetailState, } from "./callLogArtifacts"; import { getCallLogMaxEntries, getCallLogRetentionDays, getCallLogsTableMaxRows } from "../logEnv"; +import { isSqlitePagerCorruptError, notePagerCorruption } from "../db/healthCheck"; const CALL_LOG_ROTATE_THROTTLE_MS = 60_000; const CALL_LOG_ROTATE_BATCH_SIZE = 100; @@ -308,7 +309,29 @@ export function trimCallLogsToMaxRows( return { deletedRows, deletedArtifacts }; } +let callLogRotatePaused = false; + +export function isCallLogRotatePaused(): boolean { + return callLogRotatePaused; +} + +export function resetCallLogRotateFence(): void { + callLogRotatePaused = false; +} + +export function handleCallLogRotateError(error: unknown): void { + const message = error instanceof Error ? error.message : String(error); + console.error("[callLogs] Failed to rotate request artifacts:", message); + if (!isSqlitePagerCorruptError(error)) return; + callLogRotatePaused = true; + notePagerCorruption("call-log-rotate", error); + console.error( + "[callLogs] SQLITE_CORRUPT during rotation; pausing further rotate writes. Check /api/db/health." + ); +} + export function rotateCallLogs() { + if (callLogRotatePaused) return; try { if (!CALL_LOGS_DIR || !fs.existsSync(CALL_LOGS_DIR)) return; @@ -324,7 +347,7 @@ export function rotateCallLogs() { minAgeMs: CALL_LOG_ORPHAN_MIN_AGE_MS, }); } catch (error) { - console.error("[callLogs] Failed to rotate request artifacts:", (error as Error).message); + handleCallLogRotateError(error); } } @@ -335,7 +358,7 @@ function runScheduledCallLogRotation() { try { rotateCallLogs(); } catch (error) { - console.error("[callLogs] Failed to rotate request artifacts:", (error as Error).message); + handleCallLogRotateError(error); } finally { callLogRotateInFlight = false; } diff --git a/src/lib/usage/internalUsageCommand.ts b/src/lib/usage/internalUsageCommand.ts index 257b8f9c158..37f5f009d1c 100644 --- a/src/lib/usage/internalUsageCommand.ts +++ b/src/lib/usage/internalUsageCommand.ts @@ -13,7 +13,7 @@ const TEXT_PLAIN_HEADERS = { "Content-Type": "text/plain; charset=utf-8" } as co type JsonRecord = Record; -interface UsageCommandApiKeyMetadata { +export interface UsageCommandApiKeyMetadata { id: string; name?: string; allowedConnections?: string[] | null; @@ -31,7 +31,7 @@ interface ProviderConnectionLike { quotaWindowThresholds?: Record | null; } -interface UsageSnapshot { +export interface UsageSnapshot { connectionId: string; provider: string; plan: unknown; @@ -39,7 +39,7 @@ interface UsageSnapshot { quotaWindowThresholds?: Record | null; } -interface UsageCommandSelection { +export interface UsageCommandSelection { preferredProvider?: string | null; preferredConnectionId?: string | null; } @@ -258,7 +258,7 @@ function snapshotFromConnection( }; } -async function collectUsageSnapshots( +export async function collectUsageSnapshots( metadata: UsageCommandApiKeyMetadata, deps: RequiredDeps ): Promise { @@ -525,6 +525,55 @@ function appendQuotaBlock( lines.push(`⏱ reset in ${formatResetIn(getResetAt(match?.quota ?? null), now)}`); } +/** + * Structured form of the usage command — what {@link buildUsageCommandText} + * renders as text, exposed as data for API consumers (the OmniCopilot panel + * asks for it via `?format=json`). Text and JSON share the exact same + * collectors, so the two can never disagree about a number. + * + * The key design constraint is the 403 case: a key without `allowUsageCommand` + * must reach the client as a *structured* reason, not a bare text error — a + * caller rendering a usage panel has to be able to tell "the server does not + * know your limits yet" apart from "this key may not ask". + */ +/** Discriminated so the caller never reads a data field off a refusal: + * `allowed:false` carries only `error`; `allowed:true` carries the data. */ +export type UsageCommandJson = + | { allowed: false; error: { message: string } } + | { + allowed: true; + /** Present only when the key opted into per-key usage limits. */ + personal: unknown | null; + /** The selected provider snapshot, or null when nothing is cached. */ + provider: UsageSnapshot | null; + /** Every connection's snapshot, so a panel can render Codex / Claude / + * OpenCode side by side instead of only the selected one (#11191). The + * single-pick in `provider` is a presentation choice for a terminal; the + * collector already gathered all of them. */ + providers: UsageSnapshot[]; + }; + +export async function buildUsageCommandJson( + metadata: UsageCommandApiKeyMetadata, + deps: InternalUsageCommandDeps = {}, + selection: UsageCommandSelection = {} +): Promise { + const resolvedDeps = await normalizeDeps(deps); + const personal = + metadata.usageLimitEnabled === true + ? await resolvedDeps.getApiKeyUsageLimitStatus( + { + ...metadata, + preferredProvider: selection.preferredProvider ?? metadata.preferredProvider ?? null, + }, + { now: resolvedDeps.now } + ) + : null; + const snapshots = await collectUsageSnapshots(metadata, resolvedDeps); + const provider = selectUsageSnapshot(snapshots, selection); + return { allowed: true, personal, provider, providers: snapshots }; +} + export async function buildUsageCommandText( metadata: UsageCommandApiKeyMetadata, deps: InternalUsageCommandDeps = {}, @@ -588,6 +637,17 @@ function inferHttpUsageCommandSelection(request: Request): UsageCommandSelection } } +/** `?format=json` (or `?format=JSON`) — anything else falls back to the text + * form, which is the historical contract of this endpoint. */ +function wantsUsageCommandJson(request: Request): boolean { + try { + const format = new URL(request.url, "http://localhost").searchParams.get("format"); + return format !== null && format.trim().toLowerCase() === "json"; + } catch { + return false; + } +} + function createPlainUsageCommandResponse(text: string, status = 200): Response { return new Response(text, { status, headers: TEXT_PLAIN_HEADERS }); } @@ -764,22 +824,45 @@ export async function handleInternalUsageCommandHttpRequest( ): Promise { try { const resolvedDeps = await normalizeDeps(deps); + const json = wantsUsageCommandJson(request); const apiKey = extractUsageCommandApiKey(request); if (!apiKey || !(await resolvedDeps.isValidApiKey(apiKey))) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } const metadata = await resolvedDeps.getApiKeyMetadata(apiKey); if (!metadata?.id) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } if (metadata.allowUsageCommand !== true) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_DISABLED_MESSAGE } } satisfies UsageCommandJson, + { status: 403 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_DISABLED_MESSAGE, 403); } + const selection = inferHttpUsageCommandSelection(request); + if (json) { + return Response.json(await buildUsageCommandJson(metadata, resolvedDeps, selection)); + } return createPlainUsageCommandResponse( - await buildUsageCommandText(metadata, resolvedDeps, inferHttpUsageCommandSelection(request)) + await buildUsageCommandText(metadata, resolvedDeps, selection) ); } catch (err) { const body = buildErrorBody(500, err instanceof Error ? err.message : String(err)); diff --git a/src/lib/webhooks/eventDescriptions.ts b/src/lib/webhooks/eventDescriptions.ts index 00ffae98efd..c0b992a53ea 100644 --- a/src/lib/webhooks/eventDescriptions.ts +++ b/src/lib/webhooks/eventDescriptions.ts @@ -1,11 +1,11 @@ -export type WebhookEvent = - | "request.completed" - | "request.failed" - | "provider.error" - | "provider.recovered" - | "quota.exceeded" - | "combo.switched" - | "test.ping"; +export type WebhookEvent = "request.completed" | "request.failed" | "quota.exceeded" | "test.ping"; + +export const WEBHOOK_EVENT_VALUES = [ + "request.completed", + "request.failed", + "quota.exceeded", + "test.ping", +] as const; export interface EventDescription { label: string; @@ -38,35 +38,12 @@ export const EVENT_DESCRIPTIONS: Record = { attempts: 3, }, }, - "provider.error": { - label: "Provider Error", - emoji: "⚠️", - description: "A provider tripped the circuit breaker due to repeated failures.", - exampleData: { provider: "openai", model: "gpt-4o", errorCode: 503, consecutiveFailures: 3 }, - }, - "provider.recovered": { - label: "Provider Recovered", - emoji: "✅", - description: "A provider recovered from a circuit-breaker OPEN state.", - exampleData: { provider: "openai", recoveredAfterMs: 60000 }, - }, "quota.exceeded": { label: "Quota Exceeded", emoji: "📊", description: "A usage threshold (e.g. 95% of quota) was reached.", exampleData: { quota: "daily_tokens", used: 950000, limit: 1000000, pct: 95 }, }, - "combo.switched": { - label: "Combo Switched", - emoji: "🔄", - description: "Combo routing switched to a different target.", - exampleData: { - combo: "auto-fallback", - fromModel: "gpt-4o", - toModel: "claude-opus-4-7", - reason: "provider.error", - }, - }, "test.ping": { label: "Test Ping", emoji: "🏓", diff --git a/src/lib/webhooks/integrations/discord.ts b/src/lib/webhooks/integrations/discord.ts index 0b3fdc4199a..3686e427275 100644 --- a/src/lib/webhooks/integrations/discord.ts +++ b/src/lib/webhooks/integrations/discord.ts @@ -16,10 +16,7 @@ export interface DiscordPayload { const EVENT_COLORS: Partial> = { "request.completed": 0x22c55e, "request.failed": 0xef4444, - "provider.error": 0xf97316, - "provider.recovered": 0x22c55e, "quota.exceeded": 0xeab308, - "combo.switched": 0x3b82f6, "test.ping": 0x8b5cf6, }; diff --git a/src/proxy.ts b/src/proxy.ts index 153785efdf6..b99a0789d9d 100644 --- a/src/proxy.ts +++ b/src/proxy.ts @@ -24,6 +24,14 @@ export async function proxy(request: NextRequest) { return runAuthzPipeline(request, { enforce: true }); } +// Next compiles the middleware/proxy matcher from `regexp.source` only, dropping +// path-to-regexp's default case-insensitive flag — so a lowercase literal like +// `/v1/:path*` never matches `/V1/...`, while the rewrite matcher (flag kept) +// still routes it to the handler. That skipped the authz pipeline entirely +// (GHSA-jvqc-mp9f-q936). Expressing the case-insensitivity inside a custom +// path-to-regexp group (`([vV]1)`) survives the flag-drop because it needs no +// flag. Keep these in sync with the client-API aliases in +// next.config.mjs rewrites and src/server/authz/classify.ts. export const config = { matcher: [ "/", @@ -31,15 +39,15 @@ export const config = { "/home", "/home/:path*", "/api/:path*", - "/v1/:path*", - "/v1", - "/v1beta/:path*", - "/v1beta", - "/chat/:path*", - "/responses/:path*", - "/responses", - "/codex/:path*", - "/codex", - "/models", + "/:v1seg([vV]1)/:path*", + "/:v1seg([vV]1)", + "/:v1betaseg([vV]1[bB][eE][tT][aA])/:path*", + "/:v1betaseg([vV]1[bB][eE][tT][aA])", + "/:chatseg([cC][hH][aA][tT])/:path*", + "/:respseg([rR][eE][sS][pP][oO][nN][sS][eE][sS])/:path*", + "/:respseg([rR][eE][sS][pP][oO][nN][sS][eE][sS])", + "/:codexseg([cC][oO][dD][eE][xX])/:path*", + "/:codexseg([cC][oO][dD][eE][xX])", + "/:modelsseg([mM][oO][dD][eE][lL][sS])", ], }; diff --git a/src/server/authz/classify.ts b/src/server/authz/classify.ts index a5270860d62..bfe0f0d6f9d 100644 --- a/src/server/authz/classify.ts +++ b/src/server/authz/classify.ts @@ -16,30 +16,39 @@ function normalizePathname(rawPath: string): { path: string; reason?: Classifica if (!path.startsWith("/")) path = "/" + path; if (path.length > 1 && path.endsWith("/")) path = path.slice(0, -1); - if (path === "/codex" || path.startsWith("/codex/")) { + // Client-API aliases are matched case-insensitively on the control segment. + // Next's rewrite layer accepts `/V1/...`, `/CODEX`, etc. and routes them to + // the client handler, so the classifier must recognize the same casing — + // otherwise an uppercase alias falls through to the management fallback and + // the request is treated as a different route class than it is actually + // dispatched to (GHSA-jvqc-mp9f-q936). Only the leading control segment is + // lowercased for detection; the original-case tail is preserved. + const lower = path.toLowerCase(); + + if (lower === "/codex" || lower.startsWith("/codex/")) { return { path: "/api/v1/responses", reason: "client_api_codex_alias" }; } - if (path === "/v1/v1" || path.startsWith("/v1/v1/")) { + if (lower === "/v1/v1" || lower.startsWith("/v1/v1/")) { const tail = path.slice("/v1/v1".length) || ""; return { path: "/api/v1" + tail, reason: "client_api_double_prefix" }; } - if (path === "/v1beta" || path.startsWith("/v1beta/")) { + if (lower === "/v1beta" || lower.startsWith("/v1beta/")) { const tail = path.slice("/v1beta".length) || ""; return { path: "/api/v1beta" + tail, reason: "client_api_alias" }; } - if (path === "/v1" || path.startsWith("/v1/")) { + if (lower === "/v1" || lower.startsWith("/v1/")) { const tail = path.slice("/v1".length) || ""; return { path: "/api/v1" + tail, reason: "client_api_alias" }; } for (const { alias, canonical } of CLIENT_API_ALIAS_PREFIXES) { - if (path === alias) { + if (lower === alias) { return { path: canonical, reason: "client_api_alias" }; } - if (path.startsWith(alias + "/")) { + if (lower.startsWith(alias + "/")) { return { path: canonical + path.slice(alias.length), reason: "client_api_alias" }; } } diff --git a/src/server/authz/pipeline.ts b/src/server/authz/pipeline.ts index 9f4e46bed1e..d8dacf376d7 100644 --- a/src/server/authz/pipeline.ts +++ b/src/server/authz/pipeline.ts @@ -205,6 +205,7 @@ function drainingResponse(requestId: string): NextResponse { { status: 503 } ); response.headers.set(AUTHZ_HEADER_REQUEST_ID, requestId); + response.headers.set("Retry-After", "5"); return response; } diff --git a/src/server/authz/routeGuard.ts b/src/server/authz/routeGuard.ts index 4bb072d5cb1..7c61d24545c 100644 --- a/src/server/authz/routeGuard.ts +++ b/src/server/authz/routeGuard.ts @@ -56,6 +56,8 @@ export const LOCAL_ONLY_API_PREFIXES: ReadonlyArray = [ "/api/jobs", // JobRegistry control (enable/disable/run-now) + run history - runtime job administration, loopback-only (Hard Rules #15 + #17) "/api/jobs/", // sub-paths: /api/jobs/:id/{runs,enable,disable,run-now} (the bare `/api/jobs` above matches the list route; this matches children) "/api/oauth/cursor/auto-import", // spawns execFile("which", argv-array-of-one-arg "cursor") to verify a local Cursor install before importing creds — RCE-via-tunnel surface (Hard Rules #15 + #17, found by 6A.8 route-guard gate). Specific path only: the rest of /api/oauth/ (browser redirect/callback flows) must stay remote-reachable. Note: this comment intentionally avoids a literal closing square bracket character — check-openapi-security-tiers.mjs's naive regex parser for this array stops at the first one it finds, silently truncating its view of every entry after this one. + "/api/oauth/kiro/auto-import", // reads host-local Kiro credential files (homedir kiro-cli data) — must reach the loopback-only gate, not the PUBLIC /api/oauth/ prefix (GHSA-wgwc-crjm-pmwv, GHSA-gxv4-955v-v6cm). Excluded from PUBLIC in publicApiRoutes.ts. + "/api/oauth/raycast/auto-import", // reads host-local Raycast credential files — same loopback-only rationale as the kiro and cursor auto-import routes above. "/api/skills/collect/", // Skill Collector CLI detection: GET .../detect probes getCliRuntimeStatus() per CLI_TOOL_IDS entry, which spawns a child process to check each tool — RCE-via-tunnel surface (Hard Rules #15 + #17, PR #6294 review). "/api/discovery/", // Discovery tool (opt-in provider scanner): the scan route makes outbound probes to provider endpoints (SSRF-adjacent) and the whole surface is an admin research tool — strict-loopback only, no manage-scope bypass (NOT in LOCAL_ONLY_MANAGE_SCOPE_BYPASS_PREFIXES). See _tasks/features-v3.8.42/gaps/DISCOVERY_TOOL_DESIGN.md. VNC_ROUTE_PREFIX, // #7892: /api/vnc-session/* spawns Docker containers via child_process.spawn (src/lib/vncSession/service.ts) — RCE-via-tunnel surface (Hard Rules #15 + #17), same CVE class (GHSA-fhh6-4qxv-rpqj). @@ -119,6 +121,11 @@ export const ALWAYS_PROTECTED_API_PATHS: ReadonlyArray = [ "/api/shutdown", "/api/providers/health-autopilot/actions", "/api/settings/database", + // Full-database export/import: a credential dump and an irreversible replace. + // Must stay authenticated even under requireLogin=false, for the same reason + // /api/settings/database already does. isAlwaysProtectedPath matches on a path + // boundary, so this covers export, exportAll and import. (GHSA-mghq-58h3-qcqj) + "/api/db-backups", ]; export function isLoopbackHost(hostHeader: string | null): boolean { diff --git a/src/server/cors/origins.ts b/src/server/cors/origins.ts index dfd0b407c50..a354cccb5e3 100644 --- a/src/server/cors/origins.ts +++ b/src/server/cors/origins.ts @@ -144,6 +144,26 @@ export function getCorsStatus(): CorsStatus { * compression middleware only appends it conditionally, so shared caches can't * otherwise reliably tell compressed vs uncompressed variants apart. */ +function requestCarriesTokenOrPreflight(request: Request): boolean { + // Preflight (OPTIONS) never carries the Authorization / x-api-key header, so it + // must be allowed through — the actual request that follows is re-evaluated by + // this same check and only gets the permissive Origin if it presents a token. + if (request.method === "OPTIONS") return true; + if ( + request.headers.get("authorization") || + request.headers.get("x-api-key") || + request.headers.get("x-goog-api-key") + ) { + return true; + } + // A dashboard session cookie is a credential too (#5242 browser/Electron + // clients). auth_token is HttpOnly + SameSite, so a cross-site attacker page + // cannot get it auto-attached — only a truly credential-less request (the + // GHSA-7px7 anonymous case on a keyless install) falls through to fail-closed. + const cookie = request.headers.get("cookie"); + return Boolean(cookie && /(?:^|;\s*)auth_token=/.test(cookie)); +} + export function applyCorsHeaders( response: Response, request: Request, @@ -151,7 +171,15 @@ export function applyCorsHeaders( ): void { const requestOrigin = request.headers.get("origin"); let allowed = resolveAllowedOrigin(requestOrigin); - if (allowed === null && relaxForTokenAuth) { + if (allowed === null && relaxForTokenAuth && requestCarriesTokenOrPreflight(request)) { + // GHSA-7px7-29v2-m97p: the permissive Origin echo is only safe on the + // assumption that these routes are token-authenticated (browsers never + // auto-attach Authorization/x-api-key). On a keyless install that assumption + // breaks — an anonymous cross-origin page would be echoed its own Origin and + // could read the response. Only relax for a request that actually carries a + // credential, plus CORS preflights (OPTIONS never carries the header — the + // real request that follows is re-checked), so authenticated browser/Electron + // clients (#5242) keep working while credential-less cross-origin reads do not. allowed = requestOrigin && requestOrigin.length > 0 ? requestOrigin : "*"; } if (allowed !== null) { diff --git a/src/shared/components/CursorAuthModal.tsx b/src/shared/components/CursorAuthModal.tsx index 53de6a49833..b56ce066463 100644 --- a/src/shared/components/CursorAuthModal.tsx +++ b/src/shared/components/CursorAuthModal.tsx @@ -1,6 +1,6 @@ "use client"; -import { useState, useEffect } from "react"; +import { useState, useEffect, useRef } from "react"; import { useTranslations } from "next-intl"; import Modal from "./Modal"; import Button from "./Button"; @@ -13,9 +13,12 @@ type CursorAuthModalProps = { reauthConnection?: unknown; }; +type AuthTab = "login" | "import"; + /** * Cursor Auth Modal - * Auto-detect and import token from Cursor IDE's local SQLite database + * Primary: deep-control PKCE login (Docker-friendly). + * Secondary: IDE / paste-token import (optional refresh token). */ export default function CursorAuthModal({ isOpen, @@ -24,43 +27,161 @@ export default function CursorAuthModal({ reauthConnection: _, }: CursorAuthModalProps) { const t = useTranslations("cursorAuthModal"); + const [tab, setTab] = useState("login"); const [accessToken, setAccessToken] = useState(""); + const [refreshToken, setRefreshToken] = useState(""); const [machineId, setMachineId] = useState(""); - const [error, setError] = useState(null); + const [error, setError] = useState(null); const [importing, setImporting] = useState(false); const [autoDetecting, setAutoDetecting] = useState(false); const [autoDetected, setAutoDetected] = useState(false); + const [dockerHint, setDockerHint] = useState(false); + + const [loginUrl, setLoginUrl] = useState(""); + const [sessionId, setSessionId] = useState(""); + const [loginStarting, setLoginStarting] = useState(false); + const [loginPolling, setLoginPolling] = useState(false); + const pollAbortRef = useRef(false); - // Auto-detect tokens when modal opens useEffect(() => { - if (!isOpen) return; + if (!isOpen) { + pollAbortRef.current = true; + return; + } + + pollAbortRef.current = false; + + // Reset modal UI state for this open. Nested in a function (like autoDetect + // below) rather than called directly in the effect body, since these are a + // response to the modal being (re)opened — not a synchronization of React + // state with an external system — and react-hooks/set-state-in-effect flags + // direct top-level setState calls in an effect. + const resetModalState = () => { + setTab("login"); + setError(null); + setLoginUrl(""); + setSessionId(""); + setLoginPolling(false); + // Detect Docker-ish environments for import-tab guidance + setDockerHint(false); + }; + resetModalState(); const autoDetect = async () => { setAutoDetecting(true); - setError(null); setAutoDetected(false); - try { const res = await fetch("/api/oauth/cursor/auto-import"); const data = await res.json(); - if (data.found) { - setAccessToken(data.accessToken); + setAccessToken(data.accessToken || ""); + setRefreshToken(data.refreshToken || ""); setMachineId(data.machineId || ""); setAutoDetected(true); } else { - setError(data.error || t("errorAutoDetect")); + // Soft hint only on import tab; don't block login tab + if ( + typeof data.error === "string" && + /docker|config files found|not appear/i.test(data.error) + ) { + setDockerHint(true); + } } - } catch (err) { - setError(t("errorAutoDetectFailed")); + } catch { + /* ignore — login tab is primary */ } finally { setAutoDetecting(false); } }; - autoDetect(); + void autoDetect(); }, [isOpen]); + useEffect(() => { + return () => { + pollAbortRef.current = true; + }; + }, []); + + const handleStartLogin = async () => { + setLoginStarting(true); + setError(null); + setLoginUrl(""); + setSessionId(""); + try { + const res = await fetch("/api/oauth/cursor/login/start", { method: "POST" }); + const data = await res.json(); + if (!res.ok) throw new Error(data.error || t("errorLoginStart")); + setLoginUrl(data.loginUrl); + setSessionId(data.sessionId); + if (typeof window !== "undefined" && data.loginUrl) { + window.open(data.loginUrl, "_blank", "noopener,noreferrer"); + } + void pollUntilDone(data.sessionId); + } catch (err) { + setError(err instanceof Error ? err.message : t("errorLoginStart")); + } finally { + setLoginStarting(false); + } + }; + + const pollUntilDone = async (sid: string) => { + setLoginPolling(true); + pollAbortRef.current = false; + const maxAttempts = 150; + let delayMs = 1000; + try { + for (let i = 0; i < maxAttempts; i++) { + if (pollAbortRef.current) return; + await new Promise((r) => setTimeout(r, delayMs)); + if (pollAbortRef.current) return; + + const res = await fetch("/api/oauth/cursor/login/poll", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ sessionId: sid }), + }); + const data = await res.json(); + + if (data.status === "pending") { + delayMs = Math.min(delayMs * 1.2, 10_000); + continue; + } + if (data.status === "ok" || data.success) { + onSuccess?.(); + onClose(); + return; + } + throw new Error(data.error || t("errorLoginPoll")); + } + throw new Error(t("errorLoginTimeout")); + } catch (err) { + if (!pollAbortRef.current) { + setError(err instanceof Error ? err.message : t("errorLoginPoll")); + } + } finally { + setLoginPolling(false); + } + }; + + const handleCancelLogin = async () => { + pollAbortRef.current = true; + if (sessionId) { + try { + await fetch("/api/oauth/cursor/login/cancel", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ sessionId }), + }); + } catch { + /* ignore */ + } + } + setLoginPolling(false); + setLoginUrl(""); + setSessionId(""); + }; + const handleImportToken = async () => { if (!accessToken.trim()) { setError(t("errorEnterToken")); @@ -73,6 +194,7 @@ export default function CursorAuthModal({ try { const body: Record = { accessToken: accessToken.trim() }; if (machineId.trim()) body.machineId = machineId.trim(); + if (refreshToken.trim()) body.refreshToken = refreshToken.trim(); const res = await fetch("/api/oauth/cursor/import", { method: "POST", @@ -83,67 +205,122 @@ export default function CursorAuthModal({ const data = await res.json(); if (!res.ok) { - throw new Error(data.error || t("errorImportFailed")); + throw new Error( + typeof data.error === "string" + ? data.error + : data.error?.message || t("errorImportFailed") + ); } - // Success - close modal and trigger refresh onSuccess?.(); onClose(); } catch (err) { - setError(err.message); + setError(err instanceof Error ? err.message : t("errorImportFailed")); } finally { setImporting(false); } }; + const handleClose = () => { + void handleCancelLogin(); + onClose(); + }; + return ( - +
- {/* Auto-detecting state */} - {autoDetecting && ( -
-
- - progress_activity - +
+ + +
+ + {tab === "login" && ( +
+

{t("loginDescription")}

+ + {loginPolling && ( +
+
+ + progress_activity + +
+

{t("waitingApproval")}

+ {loginUrl && ( +

+ {t("openUrlHint")}{" "} + + {t("openLoginLink")} + +

+ )} +
+ )} + + {error && ( +
+

{error}

+
+ )} + +
+ {!loginPolling ? ( + + ) : ( + + )} +
-

{t("autoDetecting")}

-

{t("readingFromCursor")}

)} - {/* Form (shown after auto-detect completes) */} - {!autoDetecting && ( - <> - {/* Success message if auto-detected */} - {autoDetected && ( + {tab === "import" && ( +
+ {autoDetecting && ( +
+

{t("autoDetecting")}

+
+ )} + + {!autoDetecting && autoDetected && (
-
- - check_circle - -

- {t("tokensAutoDetected")} -

-
+

+ {t("tokensAutoDetected")} +

)} - {/* Info message if not auto-detected */} - {!autoDetected && !error && ( + {!autoDetecting && !autoDetected && (
-
- - info - -

- {t("cursorNotDetected")} -

-
+

+ {dockerHint ? t("dockerImportHint") : t("cursorNotDetected")} +

)} - {/* Access Token Input */}
- {/* Machine ID Input (optional — not needed for cursor-agent imports) */} +
+ +
InterfaceEndpoint / commandUse it for
🧰 MCP (stdio)omniroute --mcpPlug into Claude Desktop, Cursor, any MCP client
🌊 MCP (HTTP)/api/mcp/streamRemote MCP — 109 tools, 33 scopes, full audit trail
🌊 MCP (HTTP)/api/mcp/streamRemote MCP — 110 tools, 33 scopes, full audit trail
📡 MCP (SSE)/api/mcp/sseStreaming MCP transport
🤝 A2A/.well-known/agent.jsonAgent-to-agent, JSON-RPC 2.0 + SSE, 6 skills
🌐 REST API/v1/*OpenAI-compatible — chat, embeddings, images, audio, OCR
RuntimeNode.js 22.x / 24.x LTS — >=22.22.2 <23 || >=24.0.0 <27
LanguageTypeScript 6.0 — 100% TypeScript across src/ and open-sse/ (zero any in core since v2.0)
FrameworkNext.js 16 + React 19 + Tailwind CSS 4
Databasebetter-sqlite3 (SQLite, WAL journaling) + LowDB (JSON legacy) — 120 domain modules, 157 migrations
Databasebetter-sqlite3 (SQLite, WAL journaling) + LowDB (JSON legacy) — 120 domain modules, 159 migrations
MemorySQLite FTS5 full-text + int8-quantized vector embeddings, typed decay
SchemasZod 4 — MCP tool I/O validation + API contracts
ProtocolsMCP (stdio / HTTP / SSE) + A2A v0.3 (JSON-RPC 2.0 + SSE)