diff --git a/.dockerignore b/.dockerignore index 9f7dea5a316..069c1e67812 100644 --- a/.dockerignore +++ b/.dockerignore @@ -67,3 +67,42 @@ images clipr omnirouteCloud omnirouteSite + +# Temporary/Scratch Folders +_* + +# CI/CD and Version Control (that are not actual code) +.github +.husky +.omc + +# Test Configs and Reports +playwright.config.ts +vitest*.ts +audit-report.json +sonar-project.properties + +# Deployment Configs +docker-compose*.yml +fly.toml + +# Consistent with .gitignore +.DS_Store +.idea/ +.config/ +.data/ +.omnivscodeagent/ +*.sqlite-* +*.tsbuildinfo +next-env.d.ts +security-analysis/ +.analysis/ +antigravity-manager-analysis/ +.sisyphus/ +.plans/ +app.__qa_backup/ +.app-build-backup-*/ +.gitnexus +.worktrees +.next-playwright/ +cloud/ diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6778e667fa7..f8b3ff80817 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,7 +15,8 @@ permissions: contents: read env: - CI_NODE_VERSION: "22.22.2" + CI_NODE_VERSION: "24" + CI_NODE_24_VERSION: "24" jobs: lint: @@ -185,6 +186,26 @@ jobs: - run: npm run check:node-runtime - run: npm run test:unit + node-24-compat: + name: Node 24 Compatibility + runs-on: ubuntu-latest + timeout-minutes: 15 + needs: build + env: + JWT_SECRET: ci-test-secret-with-sufficient-length-for-validation + API_KEY_SECRET: ci-test-api-key-secret-long + DISABLE_SQLITE_AUTO_BACKUP: "true" + steps: + - uses: actions/checkout@v6 + - uses: actions/setup-node@v6 + with: + node-version: ${{ env.CI_NODE_24_VERSION }} + cache: npm + - run: npm ci + - run: npm run check:node-runtime + - run: npm run build + - run: npm run test:unit + test-coverage: name: Coverage runs-on: ubuntu-latest @@ -413,6 +434,7 @@ jobs: - build - package-artifact - test-unit + - node-24-compat - test-coverage - sonarqube - coverage-pr-comment diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml index c8bdb4b061e..84974b88a43 100644 --- a/.github/workflows/electron-release.yml +++ b/.github/workflows/electron-release.yml @@ -71,13 +71,11 @@ jobs: deb_ext: .deb steps: - - name: Checkout - uses: actions/checkout@v6 - - - name: Setup Node.js + - uses: actions/checkout@v6 + - name: Setup Node uses: actions/setup-node@v6 with: - node-version: 22 + node-version: 24 cache: npm - name: Cache node_modules diff --git a/.github/workflows/npm-publish.yml b/.github/workflows/npm-publish.yml index bb56d71bc7c..3f38d745708 100644 --- a/.github/workflows/npm-publish.yml +++ b/.github/workflows/npm-publish.yml @@ -38,7 +38,7 @@ permissions: packages: write env: - NPM_PUBLISH_NODE_VERSION: "22.22.2" + NPM_PUBLISH_NODE_VERSION: "24" jobs: publish: diff --git a/.gitignore b/.gitignore index c4abb01ea6f..783740a7d79 100644 --- a/.gitignore +++ b/.gitignore @@ -170,3 +170,9 @@ docs/superpowers/ # GitNexus local index .gitnexus .worktrees +bin/omniroute.mjs + +# Consistent with .dockerignore / .npmignore +.omc/ +audit-report.json +bun.lock diff --git a/.node-version b/.node-version index 2bd5a0a98a3..a45fd52cc58 100644 --- a/.node-version +++ b/.node-version @@ -1 +1 @@ -22 +24 diff --git a/.npmignore b/.npmignore index e0a5b886543..cc14c145c20 100644 --- a/.npmignore +++ b/.npmignore @@ -76,3 +76,27 @@ app/_*/ app/coverage/ app/logs/ app/tests/ + +# Consistent with .gitignore and .dockerignore +.DS_Store +.idea/ +.config/ +.data/ +.omnivscodeagent/ +.omc/ +*.sqlite-* +*.tsbuildinfo +security-analysis/ +.analysis/ +antigravity-manager-analysis/ +.sisyphus/ +.plans/ +app.__qa_backup/ +.app-build-backup-*/ +.gitnexus +.worktrees +.next-playwright/ +test-results/ +playwright-report/ +blob-report/ +coverage/ diff --git a/.nvmrc b/.nvmrc index 2bd5a0a98a3..a45fd52cc58 100644 --- a/.nvmrc +++ b/.nvmrc @@ -1 +1 @@ -22 +24 diff --git a/CHANGELOG.md b/CHANGELOG.md index 6791a35c2d5..b4f2a1d0630 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,14 +4,33 @@ --- -## [3.6.7] — 2026-04-16 +## [3.6.8] — 2026-04-16 ### ✨ New Features +- **feat(core):** Add full support for Node.js 24 LTS (Krypton) environments with continuous integration coverage (#1340) +- **feat(dashboard):** Display Antigravity credit balance in dashboard Limits & Quotas (#1338) - **feat(i18n):** Add internationalization support for combo features and dashboard components; sync translations across 31 keys (#1318) - -### 🐛 Bug Fixes - +- **feat(providers):** Add Claude Opus 4.7 to Claude Code OAuth models natively with extended context and caching (#1347) +- **feat(core):** Add stopSequences support and expand tool definitions to include Google Search capabilities +- **security:** Resolve GitHub CodeQL scan alerts and enforce deep SSRF mitigations + +### 🐛 Bug Fixes + +- **fix(db):** Prevent native module ABI load crashes from assuming database corruption and skipping databases +- **fix(db):** Increase mass-migration threshold from 5 to 50 pending migrations to protect legacy users upgrading node +- **fix(db):** Prevent migration runner safety aborts from triggering on fresh `DATA_DIR` installations by detecting new databases (#1328) +- **fix(mcp):** Checkpoint and close MCP audit SQLite database safely on process signals and shutdown (#1348) +- **fix(mcp):** Fully decouple MCP audit SQLite connection caching via globalThis to fix unhandled teardown in standalone Next.js chunks (#1349) +- **fix(cli):** Avoid creating app router directory during postinstall initialization on non-built source trees (#1351) +- **fix(codex):** Correctly translate `system` role to `developer` in input array to unlock GPT-5 automatic prompt caching (#1346) +- **fix(core):** Pass client headers to executor in chatCore (#1335) +- **fix(providers):** Separate test batch calls and ignore unknown connections +- **fix(providers):** Add grok-web SSO cookie validation handler (#1334) +- **fix(db):** Preserve key_value settings (dashboard passwords, saved aliases) across DB heuristic recreation cycles (#1333) +- **fix(routing):** Allow combo fallback to cascade context overflow 400 errors instead of immediate aborts (#1331) +- **fix(core):** Resolve thinking leaks, consecutive roles, and missing thoughtSignatures for Antigravity translator (#1316) +- **fix(providers):** Default to batch testing execution blocks for web, search, and audio modalities to prevent connection timeouts - **fix(cli):** Resolve Node 22 TS entrypoint incompatibility by using esbuild compilation (#1315) - **fix(chat):** Preserve max_output_tokens for Responses API targets in chatCore sanitization (#1313) - **fix(api):** API Manager usage stats showing 0 for all registered keys (#1310) diff --git a/Dockerfile b/Dockerfile index 1aafbfd65fb..706ba212c1d 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -FROM node:22.22.2-trixie-slim AS builder +FROM node:24.14.1-trixie-slim AS builder WORKDIR /app RUN apt-get update \ @@ -13,7 +13,7 @@ RUN if [ -f package-lock.json ]; then npm ci --no-audit --no-fund; else npm inst COPY . ./ RUN mkdir -p /app/data && npm run build -- --webpack -FROM node:22.22.2-trixie-slim AS runner-base +FROM node:24.14.1-trixie-slim AS runner-base WORKDIR /app LABEL org.opencontainers.image.title="omniroute" \ diff --git a/GEMINI.md b/GEMINI.md new file mode 100644 index 00000000000..ead803af35a --- /dev/null +++ b/GEMINI.md @@ -0,0 +1,15 @@ +# Security and Cleanliness Rules for AI Assistants + +## 1. File Placement & Organization + +- **Test Files**: ALL unit tests, integration tests, ecosystem tests, or Vitest files MUST strictly be placed within the `tests/` directory (e.g., `tests/unit/`, `tests/integration/`). NEVER create test files in the project root (`/`). +- **Scripts and Utilities**: ALL maintenance, debugging, generation, or experimental scripts (`.cjs`, `.mjs`, `.js`, `.ts`) MUST be placed strictly inside the `scripts/` directory or `scripts/scratch/` for temporary one-offs. NEVER dump loose scripts in the project root (`/`). + +**The Project Root MUST ONLY CONTAIN:** + +- Configuration files (`vitest.config.ts`, `next.config.mjs`, `eslint.config.mjs`, etc.) +- Dependency files (`package.json`, `package-lock.json`) +- Documentation files (`README.md`, `CHANGELOG.md`, `AGENTS.md`) +- CI/CD files and ignore definitions (`.gitignore`, `.dockerignore`) + +When creating _any_ validation tests or one-off logic scripts, default to using `scripts/scratch/` or the `tests/unit/` directories according to your goals. Do not pollute the `/` root context. diff --git a/README.md b/README.md index 72360615163..7f45c559c04 100644 --- a/README.md +++ b/README.md @@ -695,7 +695,7 @@ During deep debugging, long histories with tool results quickly exceed provider ```txt Combo: "maximize-claude" - 1. cc/claude-opus-4-6 + 1. cc/claude-opus-4-7 2. glm/glm-4.7 3. if/kimi-k2-thinking @@ -719,7 +719,7 @@ Outcome: stable free coding workflow ```txt Combo: "always-on" - 1. cc/claude-opus-4-6 + 1. cc/claude-opus-4-7 2. cx/gpt-5.2-codex 3. glm/glm-4.7 4. minimax/MiniMax-M2.1 @@ -1511,7 +1511,7 @@ OmniRoute v3.6 is built as an operational platform, not just a relay proxy. ```txt Combo: "my-coding-stack" - 1. cc/claude-opus-4-6 + 1. cc/claude-opus-4-7 2. nvidia/llama-3.3-70b 3. glm/glm-4.7 4. if/kimi-k2-thinking @@ -1650,7 +1650,7 @@ Dashboard → Providers → Connect Claude Code → 5-hour + weekly quota tracking Models: - cc/claude-opus-4-6 + cc/claude-opus-4-7 cc/claude-sonnet-4-5-20250929 cc/claude-haiku-4-5-20251001 ``` @@ -1850,7 +1850,7 @@ Dashboard → Combos → Create New Name: premium-coding Models: - 1. cc/claude-opus-4-6 (Subscription primary) + 1. cc/claude-opus-4-7 (Subscription primary) 2. glm/glm-4.7 (Cheap backup, $0.6/1M) 3. minimax/MiniMax-M2.1 (Cheapest fallback, $0.20/1M) @@ -1880,7 +1880,7 @@ Cost: $0 forever! Settings → Models → Advanced: OpenAI API Base URL: http://localhost:20128/v1 OpenAI API Key: [from OmniRoute dashboard] - Model: cc/claude-opus-4-6 + Model: cc/claude-opus-4-7 ``` ### Claude Code @@ -1990,7 +1990,7 @@ opencode **Rate limiting** - Subscription quota out → Fallback to GLM/MiniMax -- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Add combo: `cc/claude-opus-4-7 → glm/glm-4.7 → if/kimi-k2-thinking` **OAuth token expired** @@ -2322,9 +2322,23 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 📊 Star History -## Stargazers over time - -## [![Stargazers over time](https://starchart.cc/diegosouzapw/OmniRoute.svg?variant=adaptive)](https://starchart.cc/diegosouzapw/OmniRoute) + + + + + Star History Chart + + + +## 🌍 StarMapper + + + + + + StarMapper + + ## 🙏 Acknowledgments diff --git a/docker-compose.yml b/docker-compose.yml index 4eae86ba200..7e24dbce844 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -20,6 +20,7 @@ x-common: &common restart: unless-stopped + stop_grace_period: 40s env_file: .env environment: - DATA_DIR=/app/data # Must match the volume mount below @@ -101,7 +102,7 @@ services: # Adjust paths below to match YOUR host system. - ~/.local/bin:/host-local/bin:ro # Node global binaries (adjust node version path) - # - ~/.nvm/versions/node/v22.16.0/bin:/host-node/bin:ro + # - ~/.nvm/versions/node/v24.14.1/bin:/host-node/bin:ro # ── Host config mounts (read-write) ── - ~/.codex:/host-home/.codex:rw - ~/.claude:/host-home/.claude:rw diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 8a9c13fa62f..f9a69755bbe 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -15,7 +15,7 @@ Common problems and solutions for OmniRoute. | No logs written to disk | Set `APP_LOG_TO_FILE=true` and verify call log capture is enabled | | EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | | Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | -| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Login crash / blank page | Check Node.js version — see [Node.js Compatibility](#nodejs-compatibility) below | | `dlopen` / `slice is not valid mach-o file` (macOS) | Run `cd $(npm root -g)/omniroute/app && npm rebuild better-sqlite3 && omniroute` — see [macOS native module rebuild](#macos-native-module-rebuild) below | | Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | @@ -27,10 +27,7 @@ Common problems and solutions for OmniRoute. ### Login page crashes or shows "Module self-registration" error -**Cause:** You are running a Node.js version outside OmniRoute's approved secure runtime floor. Two cases matter: - -1. **Node.js 24+**: `better-sqlite3` is not supported here and startup can fail hard. -2. **Older Node 20/22 patch levels**: the runtime may start, but it falls below the patched security floor OmniRoute now requires. +**Cause:** You are running a Node.js version outside OmniRoute's approved secure runtime floor. The most common case is running an older Node 20, 22, or 24 patch level that falls below the patched security floor OmniRoute requires. **Symptoms:** @@ -40,16 +37,16 @@ Common problems and solutions for OmniRoute. **Fix:** -1. Install a patched Node.js 22 LTS release (recommended): +1. Install a supported Node.js LTS release (recommended: Node.js 24.x): ```bash - nvm install 22.22.2 - nvm use 22.22.2 + nvm install 24 + nvm use 24 ``` -2. Verify your version: `node --version` should show `v22.22.2` or newer on the 22.x LTS line +2. Verify your version: `node --version` should show `v24.0.0` or newer on the 24.x LTS line 3. Reinstall OmniRoute: `npm install -g omniroute` 4. Restart: `omniroute` -> **Supported secure versions:** `>=20.20.2 <21` or `>=22.22.2 <23`. Node.js 24+ is **not supported**. +> **Supported secure versions:** `>=20.20.2 <21`, `>=22.22.2 <23`, or `>=24.0.0 <25`. Node.js 24.x LTS (Krypton) is fully supported. ### macOS: `dlopen` / "slice is not valid mach-o file" @@ -64,7 +61,7 @@ Common problems and solutions for OmniRoute. - Full example: ``` -dlopen(/Users//.nvm/versions/node/v24.13.1/lib/node_modules/omniroute/app/node_modules/better-sqlite3/build/Release/better_sqlite3.node, 0x0001): tried: '...' (slice is not valid mach-o file) +dlopen(/Users//.nvm/versions/node/v24.14.1/lib/node_modules/omniroute/app/node_modules/better-sqlite3/build/Release/better_sqlite3.node, 0x0001): tried: '...' (slice is not valid mach-o file) ``` **Fix — rebuild for your local environment (no Node.js downgrade required):** @@ -75,7 +72,7 @@ npm rebuild better-sqlite3 omniroute ``` -> **Note:** This recompiles the native binding against your local Node.js version and CPU architecture, resolving the binary mismatch. The officially supported secure range is now **`>=20.20.2 <21` or `>=22.22.2 <23`** (`engines` field in `package.json`). If you are on Node.js 24, the rebuild may silence this specific startup error but other issues can still occur — moving to a patched Node.js 22 LTS release remains the recommended path. +> **Note:** This recompiles the native binding against your local Node.js version and CPU architecture, resolving the binary mismatch. The officially supported range is **`>=20.20.2 <21`, `>=22.22.2 <23`, or `>=24.0.0 <25`** (`engines` field in `package.json`). Node.js 24.x LTS (Krypton) is fully supported with `better-sqlite3` v12.x. --- diff --git a/docs/USER_GUIDE.md b/docs/USER_GUIDE.md index b4892aabb38..a639755c1d6 100644 --- a/docs/USER_GUIDE.md +++ b/docs/USER_GUIDE.md @@ -55,7 +55,7 @@ Complete guide for configuring providers, creating combos, integrating CLI tools ``` Combo: "maximize-claude" - 1. cc/claude-opus-4-6 (use subscription fully) + 1. cc/claude-opus-4-7 (use subscription fully) 2. glm/glm-4.7 (cheap backup when quota out) 3. if/kimi-k2-thinking (free emergency fallback) @@ -83,7 +83,7 @@ Quality: Production-ready models ``` Combo: "always-on" - 1. cc/claude-opus-4-6 (best quality) + 1. cc/claude-opus-4-7 (best quality) 2. cx/gpt-5.2-codex (second subscription) 3. glm/glm-4.7 (cheap, resets daily) 4. minimax/MiniMax-M2.1 (cheapest, 5h reset) @@ -121,7 +121,7 @@ Dashboard → Providers → Connect Claude Code → 5-hour + weekly quota tracking Models: - cc/claude-opus-4-6 + cc/claude-opus-4-7 cc/claude-sonnet-4-5-20250929 cc/claude-haiku-4-5-20251001 ``` @@ -230,7 +230,7 @@ Dashboard → Combos → Create New Name: premium-coding Models: - 1. cc/claude-opus-4-6 (Subscription primary) + 1. cc/claude-opus-4-7 (Subscription primary) 2. glm/glm-4.7 (Cheap backup, $0.6/1M) 3. minimax/MiniMax-M2.1 (Cheapest fallback, $0.20/1M) @@ -259,7 +259,7 @@ Cost: $0 forever! Settings → Models → Advanced: OpenAI API Base URL: http://localhost:20128/v1 OpenAI API Key: [from omniroute dashboard] - Model: cc/claude-opus-4-6 + Model: cc/claude-opus-4-7 ``` ### Claude Code @@ -313,7 +313,7 @@ Edit `~/.openclaw/openclaw.json`: Provider: OpenAI Compatible Base URL: http://localhost:20128/v1 API Key: [from dashboard] -Model: cc/claude-opus-4-6 +Model: cc/claude-opus-4-7 ``` --- @@ -552,7 +552,7 @@ For the full environment variable reference, see the [README](../README.md).
View all available models -**Claude Code (`cc/`)** — Pro/Max: `cc/claude-opus-4-6`, `cc/claude-sonnet-4-5-20250929`, `cc/claude-haiku-4-5-20251001` +**Claude Code (`cc/`)** — Pro/Max: `cc/claude-opus-4-7`, `cc/claude-sonnet-4-5-20250929`, `cc/claude-haiku-4-5-20251001` **Codex (`cx/`)** — Plus/Pro: `cx/gpt-5.2-codex`, `cx/gpt-5.1-codex-max` @@ -745,7 +745,7 @@ Define global fallback chains that apply across all requests: ``` Chain: production-fallback - 1. cc/claude-opus-4-6 + 1. cc/claude-opus-4-7 2. gh/gpt-5.1-codex 3. glm/glm-4.7 ``` diff --git a/docs/openapi.yaml b/docs/openapi.yaml index d608323f5a0..93e916306c5 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -1,7 +1,7 @@ openapi: 3.1.0 info: title: OmniRoute API - version: 3.6.7 + version: 3.6.8 description: | OmniRoute is a local-first AI API proxy router. It provides an OpenAI-compatible endpoint that routes requests to multiple AI providers with load balancing, diff --git a/electron/package.json b/electron/package.json index 6ee0649e069..f37885dae66 100644 --- a/electron/package.json +++ b/electron/package.json @@ -1,6 +1,6 @@ { "name": "omniroute-desktop", - "version": "3.6.7", + "version": "3.6.8", "description": "OmniRoute Desktop Application", "main": "main.js", "author": { diff --git a/llm.txt b/llm.txt index 8df769de43f..b14bae387e0 100644 --- a/llm.txt +++ b/llm.txt @@ -8,7 +8,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo **Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. -**Current version:** 3.6.6 +**Current version:** 3.6.8 ## Tech Stack @@ -279,7 +279,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo └── .env.example # Environment variable template ``` -## Key Features (v3.6.6) +## Key Features (v3.6.8) ### Core Proxy - **60+ AI providers** with automatic format translation diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts index 7ddaa5009c8..02369d2eb5f 100644 --- a/open-sse/config/providerRegistry.ts +++ b/open-sse/config/providerRegistry.ts @@ -277,6 +277,7 @@ export const REGISTRY: Record = { tokenUrl: "https://console.anthropic.com/v1/oauth/token", }, models: [ + { id: "claude-opus-4-7", name: "Claude Opus 4.7" }, { id: "claude-opus-4-6", name: "Claude Opus 4.6" }, { id: "claude-sonnet-4-6", name: "Claude 4.6 Sonnet" }, { id: "claude-opus-4-5-20251101", name: "Claude 4.5 Opus" }, diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index d56403b98d1..83eeabe720c 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -7,8 +7,11 @@ import { classify429, decide429, type Decision } from "../services/antigravity42 import { injectCreditsField, shouldRetryWithCredits, + shouldUseCreditsFirst, + getCreditsMode, handleCreditsFailure, } from "../services/antigravityCredits.ts"; +import { persistCreditBalance, getAllPersistedCreditBalances } from "@/lib/db/creditBalance"; import { obfuscateSensitiveWords } from "../services/antigravityObfuscation.ts"; const MAX_RETRY_AFTER_MS = 60_000; @@ -28,11 +31,30 @@ const creditsExhaustedUntil = new Map(); * Per-account GOOGLE_ONE_AI remaining credit balance cache. * Populated from the final SSE chunk's `remainingCredits` field after every * successful credit-injected request. Keyed by accountId. + * On first access, hydrated from the DB-persisted balances so values survive restarts. */ const creditBalanceCache = new Map(); +let creditCacheHydrated = false; + +function hydrateCreditCacheFromDb(): void { + if (creditCacheHydrated) return; + creditCacheHydrated = true; + try { + const persisted = getAllPersistedCreditBalances(); + for (const [accountId, balance] of persisted) { + // Only fill in accounts not already populated by a live SSE response + if (!creditBalanceCache.has(accountId)) { + creditBalanceCache.set(accountId, balance); + } + } + } catch { + // DB not ready yet (build phase, etc.) — ignore silently + } +} /** Read the last-known GOOGLE_ONE_AI credit balance for a given account. */ export function getAntigravityRemainingCredits(accountId: string): number | null { + hydrateCreditCacheFromDb(); const balance = creditBalanceCache.get(accountId); return balance !== undefined ? balance : null; } @@ -40,6 +62,12 @@ export function getAntigravityRemainingCredits(accountId: string): number | null /** Update the balance cache — called when we parse `remainingCredits` from an SSE stream. */ export function updateAntigravityRemainingCredits(accountId: string, balance: number): void { creditBalanceCache.set(accountId, balance); + // Persist to DB so the value survives server restarts + try { + persistCreditBalance(accountId, balance); + } catch { + // Non-critical — in-memory cache is the primary source + } } function isCreditsExhausted(accountId: string): boolean { @@ -150,13 +178,27 @@ export class AntigravityExecutor extends BaseExecutor { // Antigravity rejects synthetic thought text, but Gemini 3+ requires any // returned thoughtSignature metadata to survive model tool-call turns. const parts = - c.parts?.filter((p) => !p.thought && (hasFunctionCall || !p.thoughtSignature)) || []; + c.parts?.filter((p) => { + // Drop empty text parts + if (typeof p.text === "string" && p.text === "") return false; + // Drop empty functionCalls + if (p.functionCall && !p.functionCall.name) return false; + + return !p.thought && (hasFunctionCall || !p.thoughtSignature); + }) || []; return { ...c, role, parts }; }) || []; - const contents = normalizedContents.filter((c) => - Array.isArray(c.parts) ? c.parts.length > 0 : true - ); + // Merge consecutive same-role entries and filter out empty sequences + const contents = []; + for (const c of normalizedContents) { + if (!Array.isArray(c.parts) || c.parts.length === 0) continue; + if (contents.length > 0 && contents[contents.length - 1].role === c.role) { + contents[contents.length - 1].parts.push(...c.parts); + } else { + contents.push(c); + } + } const transformedRequest = { ...body.request, @@ -412,17 +454,31 @@ export class AntigravityExecutor extends BaseExecutor { // non-streaming Response so chatCore's non-streaming path stays unchanged. const upstreamStream = true; - // Account ID for credits-exhausted tracking. - // Key must match getAntigravityUsage() in fetcher.ts (providerSpecificData?.email || sub). - // credentials.email and credentials.sub are populated from the same OAuth token store, - // so the cache keys written here and read in the fetcher will always match. - const accountId: string = credentials?.email || credentials?.sub || "unknown"; + // Account ID for credits tracking. + // Use connectionId as the stable cache key — it's available in both the executor + // (via credentials.connectionId) and the usage fetcher (via connection.id). + // The email-based key was unreliable because email isn't always on the credentials object. + const accountId: string = credentials?.connectionId || "unknown"; + + // Resolve credits mode once per execute() call. "always" injects + // enabledCreditTypes: ["GOOGLE_ONE_AI"] on the first request so the + // preflight normal call is skipped entirely. + const creditsMode = getCreditsMode(); + const useCreditsFirst = shouldUseCreditsFirst(credentials?.accessToken || "", creditsMode); for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) { const url = this.buildUrl(model, upstreamStream, urlIndex); const headers = this.buildHeaders(credentials, upstreamStream); mergeUpstreamExtraHeaders(headers, upstreamExtraHeaders); - const transformedBody = await this.transformRequest(model, body, upstreamStream, credentials); + let transformedBody = await this.transformRequest(model, body, upstreamStream, credentials); + + // Credits-first: inject GOOGLE_ONE_AI upfront so we never try the normal + // quota path. If credits are exhausted / disabled shouldUseCreditsFirst() + // returns false and we fall back to the legacy retry-on-429 flow. + if (useCreditsFirst) { + transformedBody = injectCreditsField(transformedBody); + log?.debug?.("AG_CREDITS", "Credits-first enabled (ANTIGRAVITY_CREDITS=always)"); + } // Initialize retry counter for this URL if (!retryAttemptsByUrl[urlIndex]) { @@ -457,17 +513,29 @@ export class AntigravityExecutor extends BaseExecutor { // 1. Try to parse explicit retry time from message const parsedRetryMs = this.parseRetryFromErrorMessage(errorMessage); - // 2. Classify 429 - const category = classify429(errorMessage); + // 2. Classify 429 (pass header-parsed retry hint as fallback + // signal — multi-hour Retry-After upgrades rate_limited to + // quota_exhausted so the GOOGLE_ONE_AI credits retry fires). + const effectiveRetryHintMs = retryMs ?? parsedRetryMs ?? null; + const category = classify429(errorMessage, effectiveRetryHintMs); // 3. For quota_exhausted, attempt Google One AI credits retry FIRST! + // Skip if credits were already injected on the first call + // (creditsMode === "always") — no point re-running with the + // same body. Record the failure so the 5h breaker kicks in. + const creditsAlreadyInjected = + (transformedBody as { enabledCreditTypes?: unknown }).enabledCreditTypes != null; + + if (category === "quota_exhausted" && creditsAlreadyInjected) { + handleCreditsFailure(credentials?.accessToken || ""); + log?.warn?.("AG_CREDITS", "Credits-first request 429'd — credits likely exhausted"); + markCreditsExhausted(accountId); + } + if ( category === "quota_exhausted" && - shouldRetryWithCredits( - credentials?.accessToken || "", - process.env.ANTIGRAVITY_CREDITS === "1" || - process.env.ANTIGRAVITY_CREDITS === "true" - ) + !creditsAlreadyInjected && + shouldRetryWithCredits(credentials?.accessToken || "", creditsMode !== "off") ) { log?.info?.("AG_CREDITS", "Retrying with Google One AI credits"); const creditsBody = injectCreditsField(transformedBody); @@ -613,7 +681,7 @@ export class AntigravityExecutor extends BaseExecutor { // For non-streaming clients, collect the SSE stream and return a synthetic // non-streaming Response so chatCore doesn't need to handle SSE conversion. if (!stream) { - return this.collectStreamToResponse( + const collected = await this.collectStreamToResponse( response, model, url, @@ -622,6 +690,82 @@ export class AntigravityExecutor extends BaseExecutor { log, signal ); + // When credits were injected (credits-first or credits-retry), the + // synthetic body contains _remainingCredits — mirror it into the + // balance cache so the dashboard stays fresh. + try { + const syntheticJson = await collected.response.clone().json(); + const rc = syntheticJson?._remainingCredits; + if (Array.isArray(rc)) { + const googleCredit = rc.find( + (c: { creditType?: string }) => c?.creditType === "GOOGLE_ONE_AI" + ); + if (googleCredit) { + const balance = parseInt(googleCredit.creditAmount, 10); + if (!isNaN(balance)) updateAntigravityRemainingCredits(accountId, balance); + } + } + } catch { + /* balance cache is best-effort */ + } + return collected; + } + + // Streaming path: wrap the response body in a pass-through TransformStream + // that extracts remainingCredits from the final SSE chunk(s) without + // consuming the stream. The client receives the unmodified SSE data. + if (response.body) { + let sseBuffer = ""; + const passThrough = new TransformStream({ + transform(chunk, controller) { + controller.enqueue(chunk); + // Accumulate text to scan for remainingCredits + try { + const text = new TextDecoder().decode(chunk, { stream: true }); + sseBuffer += text; + } catch { + /* decoding best-effort */ + } + }, + flush() { + // Parse the accumulated SSE data for remainingCredits + try { + const lines = sseBuffer.split("\n"); + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + try { + const parsed = JSON.parse(payload); + if (Array.isArray(parsed?.remainingCredits)) { + const googleCredit = parsed.remainingCredits.find( + (c) => c?.creditType === "GOOGLE_ONE_AI" + ); + if (googleCredit) { + const balance = parseInt(googleCredit.creditAmount, 10); + if (!isNaN(balance)) { + updateAntigravityRemainingCredits(accountId, balance); + } + } + } + } catch { + /* skip malformed lines */ + } + } + } catch { + /* credits extraction is best-effort */ + } + sseBuffer = ""; + }, + }); + const tappedBody = response.body.pipeThrough(passThrough); + const tappedResponse = new Response(tappedBody, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }); + return { response: tappedResponse, url, headers, transformedBody }; } return { response, url, headers, transformedBody }; diff --git a/open-sse/executors/base.ts b/open-sse/executors/base.ts index ee953afe10f..c623802d7df 100644 --- a/open-sse/executors/base.ts +++ b/open-sse/executors/base.ts @@ -390,6 +390,7 @@ export class BaseExecutor { // Only supported for specific Claude models per Anthropic docs if (extendedContext) { const EXTENDED_CONTEXT_MODELS = [ + "claude-opus-4-7", "claude-opus-4-6", "claude-sonnet-4-6", "claude-sonnet-4-5", diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index 9a631bcfda7..e04f7d172f5 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -224,6 +224,39 @@ function hoistSystemMessagesToInstructions(body: Record): void body.input = filteredInput; } +/** + * Convert role=system messages in `input` to role=developer. + * + * GPT-5 models support the `developer` role in input, but reject `system`. + * Unlike hoistSystemMessagesToInstructions(), this keeps the content inside + * the `input` array where it benefits from OpenAI's automatic prompt caching. + * + * OpenAI's prompt caching matches on the serialized prefix of the `input` array + * (+ tools). The `instructions` field is NOT included in the cache key for + * GPT-5 models. Moving system prompts from `input` to `instructions` therefore + * removes them from the cacheable prefix, resulting in 0% cache hit rates. + * + * Ref: https://community.openai.com/t/caching-is-borked-for-gpt-5-models/1359574 + * Ref: https://community.openai.com/t/no-caching-with-model-responses/1338627 + */ +function convertSystemToDeveloperRole(body: Record): void { + if (!Array.isArray(body.input)) return; + + for (const itemValue of body.input) { + if (!itemValue || typeof itemValue !== "object" || Array.isArray(itemValue)) { + continue; + } + + const item = itemValue as Record; + const role = typeof item.role === "string" ? item.role : ""; + const type = typeof item.type === "string" ? item.type : ""; + const isSystemMessage = role === "system" && (!type || type === "message"); + if (isSystemMessage) { + item.role = "developer"; + } + } +} + function normalizeCodexTools(body: Record): void { if (!Array.isArray(body.tools)) return; @@ -436,11 +469,47 @@ export class CodexExecutor extends BaseExecutor { body.service_tier = requestDefaults.serviceTier; } - // If no instructions provided, inject default Codex instructions - // NOTE: must run before the passthrough return — Codex upstream rejects - // requests without instructions even when the body is forwarded as-is. - if (!body.instructions || body.instructions.trim() === "") { - body.instructions = CODEX_DEFAULT_INSTRUCTIONS; + // ── System prompt handling: cache-aware strategy ── + // + // For GPT-5 models, OpenAI's automatic prompt caching only considers the + // `input` array content (+ tools). The `instructions` field is NOT included + // in the cache prefix computation. Moving system prompts from `input` into + // `instructions` therefore removes them from the cacheable prefix, causing + // 0% cache hit rates even with identical repeated requests. + // + // For native passthrough (client sends Responses API format directly): + // - Convert system → developer role in-place (Codex accepts developer but rejects system) + // - Only inject minimal instructions if the field is completely empty + // - Do NOT inject CODEX_DEFAULT_INSTRUCTIONS (it would bloat the non-cached field) + // + // For translated requests (from Chat Completions format): + // - Continue hoisting system messages to instructions (legacy behavior) + // - Inject CODEX_DEFAULT_INSTRUCTIONS as fallback + // + // Ref: https://community.openai.com/t/caching-is-borked-for-gpt-5-models/1359574 + // Ref: https://community.openai.com/t/no-caching-with-model-responses/1338627 + if (nativeCodexPassthrough) { + // Passthrough path: keep system prompts in input for caching. + // Convert system → developer role since Codex rejects role=system in input. + convertSystemToDeveloperRole(body); + + // Codex still requires a non-empty instructions field. + // Use a minimal placeholder if the client didn't provide one. + if ( + !body.instructions || + (typeof body.instructions === "string" && body.instructions.trim() === "") + ) { + body.instructions = "Follow the developer instructions in the conversation."; + } + } else { + // Translated path: hoist system messages to instructions (legacy behavior). + if ( + !body.instructions || + (typeof body.instructions === "string" && body.instructions.trim() === "") + ) { + body.instructions = CODEX_DEFAULT_INSTRUCTIONS; + } + hoistSystemMessagesToInstructions(body); } if (!storeEnabled) { @@ -449,10 +518,6 @@ export class CodexExecutor extends BaseExecutor { body.store = responsesStoreMarker; } - // Cursor can send native Responses payloads with role=system items inside `input`. - // Codex rejects system messages there; they must be folded into `instructions`. - hoistSystemMessagesToInstructions(body); - // Codex Responses only supports function tools with non-empty names. // Cursor may include custom tools (e.g. ApplyPatch) that work locally but are // invalid upstream, and translation bugs can leave orphaned/empty tool_choice names. diff --git a/open-sse/executors/cursor.ts b/open-sse/executors/cursor.ts index c3faf2b2905..e914d6a0182 100644 --- a/open-sse/executors/cursor.ts +++ b/open-sse/executors/cursor.ts @@ -1,4 +1,5 @@ declare const EdgeRuntime: string | undefined; +import crypto from "node:crypto"; /** * CursorExecutor — Handles communication with the Cursor IDE API. * @@ -30,7 +31,6 @@ import { import { estimateUsage } from "../utils/usageTracking.ts"; import { getCursorVersion } from "../utils/cursorVersionDetector.ts"; import { FORMATS } from "../translator/formats.ts"; -import crypto from "crypto"; import { v5 as uuidv5 } from "uuid"; import zlib from "zlib"; diff --git a/open-sse/executors/perplexity-web.ts b/open-sse/executors/perplexity-web.ts index b0232dd5c23..6b1c287d7d5 100644 --- a/open-sse/executors/perplexity-web.ts +++ b/open-sse/executors/perplexity-web.ts @@ -6,6 +6,7 @@ * completions format and Perplexity's internal protocol. */ +import crypto from "node:crypto"; import { BaseExecutor, type ExecuteInput } from "./base.ts"; const PPLX_SSE_ENDPOINT = "https://www.perplexity.ai/rest/sse/perplexity_ask"; @@ -33,9 +34,9 @@ const CITATION_RE = /\[\d+\]/g; const GROK_TAG_RE = /]*>.*?<\/grok:[^>]*>/gs; const GROK_SELF_RE = /]*\/>/g; const XML_DECL_RE = /<[?]xml[^?]*[?]>/g; -const SCRIPT_RE = /]*>.*?<\/script>/gs; -const SCRIPT_TAG_RE = /<\/?script[^>]*>/g; -const RESPONSE_TAG_RE = /<\/?response[^>]*>/g; +const SCRIPT_RE = /]*>.*?<\/script>/gis; +const SCRIPT_TAG_RE = /<\/?script\b[^>]*>/gi; +const RESPONSE_TAG_RE = /<\/?response\b[^>]*>/gi; const MULTI_SPACE = / {2,}/g; const MULTI_NL = /\n{3,}/g; @@ -109,8 +110,8 @@ function cleanResponse(text: string, strip = true): string { t = t.replace(GROK_TAG_RE, ""); t = t.replace(GROK_SELF_RE, ""); t = t.replace(RESPONSE_TAG_RE, ""); - t = t.replace(SCRIPT_RE, ""); - t = t.replace(SCRIPT_TAG_RE, ""); + t = t.replace(SCRIPT_RE, ""); // lgtm[js/incomplete-multi-character-sanitization] + t = t.replace(SCRIPT_TAG_RE, ""); // lgtm[js/incomplete-multi-character-sanitization] if (strip) { t = t.replace(MULTI_SPACE, " "); t = t.replace(MULTI_NL, "\n\n"); @@ -300,16 +301,24 @@ function buildQuery(parsed: ParsedMessages, followUpUuid: string | null): string "You have built-in web search. Answer questions directly using search results.", ]; } + + const MAX_HISTORY_ITEMS = 50; if (parsed.history.length > 0) { - obj.history = parsed.history; + obj.history = parsed.history.slice(-MAX_HISTORY_ITEMS); } + if (parsed.currentMsg) { obj.query = parsed.currentMsg; } else if (parsed.history.length === 0) { obj.query = ""; } + const json = JSON.stringify(obj); - return json.length > 96000 ? json.slice(-96000) : json; + if (json.length > 96000 && obj.history && Array.isArray(obj.history)) { + obj.history = (obj.history as any[]).slice(-10); + return JSON.stringify(obj); + } + return json; } // ─── Content extraction ───────────────────────────────────────────────────── diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 3f796d71c54..09a0f0552bf 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -1378,12 +1378,20 @@ export async function handleChatCore({ } } + // ── Proactive Context Compression (Phase 4) ── + // Check if context exceeds 85% of limit and compress proactively before sending to provider. + // This prevents "prompt too long" errors for large-but-not-full contexts. if (translatedBody && translatedBody.messages && Array.isArray(translatedBody.messages)) { const estimatedTokens = estimateTokens(JSON.stringify(translatedBody.messages)); const contextLimit = getTokenLimit(provider, effectiveModel); const COMPRESSION_THRESHOLD = 0.85; const threshold = Math.floor(contextLimit * COMPRESSION_THRESHOLD); + log?.debug?.( + "CONTEXT", + `Checking compression: ${estimatedTokens} tokens vs ${threshold} threshold (${contextLimit} limit)` + ); + if (estimatedTokens > threshold) { log?.info?.( "CONTEXT", @@ -1394,6 +1402,7 @@ export async function handleChatCore({ provider, model: effectiveModel, maxTokens: contextLimit, + reserveTokens: 0, }); if (compressionResult.compressed) { @@ -1421,8 +1430,15 @@ export async function handleChatCore({ layers: "layers" in stats ? stats.layers : undefined, }, }); + } else { + log?.debug?.("CONTEXT", `Compression not applied: context already fits within target`); } } + } else { + log?.debug?.( + "CONTEXT", + `Skipping compression check: translatedBody=${!!translatedBody}, messages=${!!translatedBody?.messages}, isArray=${Array.isArray(translatedBody?.messages)}` + ); } // Resolve executor with optional upstream proxy (CLIProxyAPI) routing. @@ -1563,6 +1579,7 @@ export async function handleChatCore({ log, extendedContext, upstreamExtraHeaders: buildUpstreamHeadersForExecute(modelToCall), + clientHeaders: clientRawRequest?.headers ?? null, }); // Qwen 429 strict quota backoff (wait 1.5s, 3s and retry) @@ -1763,6 +1780,7 @@ export async function handleChatCore({ log, extendedContext, upstreamExtraHeaders: buildUpstreamHeadersForExecute(retryModelId), + clientHeaders: clientRawRequest?.headers ?? null, }); if (retryResult.response.ok) { diff --git a/open-sse/index.ts b/open-sse/index.ts index bc7e49559db..fffb5ea17e4 100644 --- a/open-sse/index.ts +++ b/open-sse/index.ts @@ -59,7 +59,7 @@ export { refreshGoogleToken, refreshQwenToken, refreshCodexToken, - refreshIflowToken, + refreshQoderToken, refreshGitHubToken, refreshCopilotToken, getAccessToken, diff --git a/open-sse/mcp-server/__tests__/audit.test.ts b/open-sse/mcp-server/__tests__/audit.test.ts new file mode 100644 index 00000000000..90e5713a589 --- /dev/null +++ b/open-sse/mcp-server/__tests__/audit.test.ts @@ -0,0 +1,89 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; + +type MockAuditDb = { + prepare: ReturnType; + pragma: ReturnType; + close: ReturnType; + open?: boolean; +}; + +function createStatementMock() { + return { + get: vi.fn(), + all: vi.fn(), + run: vi.fn(), + }; +} + +describe("MCP audit shutdown", () => { + let dataDir: string; + let dbFile: string; + + beforeEach(() => { + vi.resetModules(); + globalThis.__omnirouteMcpAuditDb = undefined; + dataDir = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-mcp-audit-")); + dbFile = path.join(dataDir, "storage.sqlite"); + fs.writeFileSync(dbFile, ""); + process.env.DATA_DIR = dataDir; + }); + + afterEach(() => { + delete process.env.DATA_DIR; + globalThis.__omnirouteMcpAuditDb = undefined; + vi.restoreAllMocks(); + }); + + it("checkpoints and closes the audit database during shutdown", async () => { + const mockDb: MockAuditDb = { + prepare: vi.fn(() => createStatementMock()), + pragma: vi.fn(), + close: vi.fn(), + open: true, + }; + const MockDatabase = vi.fn(function MockDatabase() { + return mockDb; + }); + + vi.doMock("better-sqlite3", () => ({ + default: MockDatabase, + })); + + const audit = await import("../audit.ts"); + + await audit.logToolCall("omniroute_get_health", { ok: true }, { ok: true }, 12, true); + expect(mockDb.prepare).toHaveBeenCalledTimes(1); + + expect(audit.closeAuditDb()).toBe(true); + expect(mockDb.pragma).toHaveBeenCalledWith("wal_checkpoint(TRUNCATE)"); + expect(mockDb.close).toHaveBeenCalledTimes(1); + expect(audit.closeAuditDb()).toBe(false); + }); + + it("still closes the audit database when checkpoint fails", async () => { + const mockDb: MockAuditDb = { + prepare: vi.fn(() => createStatementMock()), + pragma: vi.fn(() => { + throw new Error("database is busy"); + }), + close: vi.fn(), + open: true, + }; + const MockDatabase = vi.fn(function MockDatabase() { + return mockDb; + }); + + vi.doMock("better-sqlite3", () => ({ + default: MockDatabase, + })); + + const audit = await import("../audit.ts"); + + await audit.logToolCall("omniroute_get_health", {}, {}, 5, true); + expect(audit.closeAuditDb()).toBe(true); + expect(mockDb.close).toHaveBeenCalledTimes(1); + }); +}); diff --git a/open-sse/mcp-server/audit.ts b/open-sse/mcp-server/audit.ts index 211c7fa143a..18897035e06 100644 --- a/open-sse/mcp-server/audit.ts +++ b/open-sse/mcp-server/audit.ts @@ -18,6 +18,13 @@ interface StatementLike { interface AuditDatabase { prepare: (sql: string) => StatementLike; + pragma: (sql: string) => unknown; + close: () => void; + open?: boolean; +} + +declare global { + var __omnirouteMcpAuditDb: AuditDatabase | null | undefined; } interface AuditStatsRow { @@ -121,7 +128,13 @@ function buildAuditFilterSql(filters: McpAuditQuery): { whereSql: string; params }; } -let db: AuditDatabase | null = null; +function getCachedAuditDb(): AuditDatabase | null { + return globalThis.__omnirouteMcpAuditDb ?? null; +} + +function setCachedAuditDb(database: AuditDatabase | null): void { + globalThis.__omnirouteMcpAuditDb = database; +} function toNumber(value: unknown, fallback = 0): number { const parsed = @@ -142,7 +155,8 @@ function toString(value: unknown): string { * Uses the same SQLite database as the main OmniRoute app. */ async function getDb(): Promise { - if (db) return db; + const cachedDb = getCachedAuditDb(); + if (cachedDb) return cachedDb; try { // Try importing the db module from the main app @@ -162,8 +176,9 @@ async function getDb(): Promise { const Database = (await import("better-sqlite3")).default as unknown as new ( dbPath: string ) => AuditDatabase; - db = new Database(dbPath); - return db; + const database = new Database(dbPath); + setCachedAuditDb(database); + return database; } catch (err: unknown) { const message = err instanceof Error ? err.message : String(err); console.error("[MCP Audit] Failed to connect to database:", message); @@ -171,6 +186,33 @@ async function getDb(): Promise { } } +export function closeAuditDb(): boolean { + const database = getCachedAuditDb(); + if (!database) return false; + + setCachedAuditDb(null); + + try { + try { + if (database.open !== false) { + database.pragma("wal_checkpoint(TRUNCATE)"); + } + } catch (err: unknown) { + const message = err instanceof Error ? err.message : String(err); + console.warn("[MCP Audit] WAL checkpoint failed during close:", message); + } + } finally { + try { + database.close(); + } catch (err: unknown) { + const message = err instanceof Error ? err.message : String(err); + console.warn("[MCP Audit] Failed to close database:", message); + } + } + + return true; +} + // ============ Audit Logger ============ /** diff --git a/open-sse/mcp-server/server.ts b/open-sse/mcp-server/server.ts index d716d2c3fd1..b158bc4277e 100644 --- a/open-sse/mcp-server/server.ts +++ b/open-sse/mcp-server/server.ts @@ -43,7 +43,7 @@ import { } from "./schemas/tools.ts"; import { startMcpHeartbeat } from "./runtimeHeartbeat.ts"; -import { logToolCall } from "./audit.ts"; +import { closeAuditDb, logToolCall } from "./audit.ts"; import { evaluateToolScopes, resolveCallerScopeContext, @@ -876,6 +876,9 @@ export async function startMcpStdio(): Promise { await server.connect(transport); console.error("[MCP] OmniRoute MCP Server connected and ready."); } finally { + if (closeAuditDb()) { + console.error("[MCP] Audit database checkpointed and closed."); + } stopHeartbeatOnce(); process.off("exit", stopHeartbeatOnce); process.off("SIGINT", stopHeartbeatOnce); diff --git a/open-sse/package.json b/open-sse/package.json index 643ea73fd34..9fbfe86a8fa 100644 --- a/open-sse/package.json +++ b/open-sse/package.json @@ -1,6 +1,6 @@ { "name": "@omniroute/open-sse", - "version": "3.6.7", + "version": "3.6.8", "description": "Express SSE sidecar for OmniRoute — handles streaming, protocol translation, and provider orchestration", "type": "module", "main": "index.js", diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index cfd621dcaa4..95bdf510df7 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -61,6 +61,31 @@ export const OAUTH_INVALID_TOKEN_SIGNALS = [ "invalid credentials", ]; +// Context overflow patterns — the prompt exceeds the model's maximum context length. +// Different providers phrase this differently. Used to decide whether a 400 error +// should trigger combo fallback (a different model may have a larger context window). +const CONTEXT_OVERFLOW_PATTERNS = [ + /\binput is too long\b/i, + /\binput too long\b/i, + /\bcontext.*(too long|exceeded|overflow|limit)/i, + /\btoo many tokens\b/i, + /\bprompt is too long\b/i, + /\bcontext window/i, + /\bmaximum context/i, + /\bmax.*token/i, + /\btoken limit/i, + /\brequest too large\b/i, +]; + +// Malformed request patterns — the model rejected the message format but a different +// provider/model in the combo may accept it. +const MALFORMED_REQUEST_PATTERNS = [ + /\bimproperly formed request\b/i, + /\binvalid.*message.*format/i, + /\bmessages must alternate/i, + /\bempty (message|content)/i, +]; + /** * T06: Returns true if response body indicates the account is permanently deactivated. */ @@ -860,8 +885,20 @@ export function checkFallbackError( }; } - // 400 Bad Request - don't fallback (same request will fail on all accounts) + // 400 — context overflow / malformed request may succeed on another model in the combo if (status === HTTP_STATUS.BAD_REQUEST) { + const isOverflow = CONTEXT_OVERFLOW_PATTERNS.some((p) => p.test(errorStr)); + const isMalformed = MALFORMED_REQUEST_PATTERNS.some((p) => p.test(errorStr)); + + if (isOverflow || isMalformed) { + return { + shouldFallback: true, + cooldownMs: 0, + reason: RateLimitReason.MODEL_CAPACITY, + }; + } + + // Generic 400 — same request will likely fail on all accounts; don't fallback. return { shouldFallback: false, cooldownMs: 0, reason: RateLimitReason.UNKNOWN }; } diff --git a/open-sse/services/antigravityCredits.ts b/open-sse/services/antigravityCredits.ts index 31cd7449837..b56bcb7eed4 100644 --- a/open-sse/services/antigravityCredits.ts +++ b/open-sse/services/antigravityCredits.ts @@ -40,3 +40,29 @@ export function shouldRetryWithCredits(authKey: string, creditsEnabled: boolean) export function handleCreditsFailure(authKey: string): boolean { return recordCreditsFailure(authKey); } + +/** + * Read the ANTIGRAVITY_CREDITS env var to determine the credits injection strategy. + * + * - "off" — never inject credits (default if env var is missing) + * - "retry" — inject credits only as a 429 fallback + * - "always" — inject credits on every request (skip normal quota path) + */ +export type CreditsMode = "off" | "retry" | "always"; + +export function getCreditsMode(): CreditsMode { + const raw = (process.env.ANTIGRAVITY_CREDITS || "").trim().toLowerCase(); + if (raw === "always" || raw === "retry") return raw; + return "off"; +} + +/** + * Determine if the executor should inject credits on the *first* request + * (credits-first mode). Returns true only when creditsMode === "always" + * and the auth key hasn't been disabled by repeated failures. + */ +export function shouldUseCreditsFirst(authKey: string, creditsMode: CreditsMode | string): boolean { + if (creditsMode !== "always") return false; + if (isCreditsDisabled(authKey)) return false; + return true; +} diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index d0434d494c5..8062b108b37 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -1448,6 +1448,7 @@ export async function handleComboChat({ percentUsed: quotaInfo.percentUsed, messages: handoffSourceMessages, model: modelStr, + comboTargets: orderedTargets.map((t) => t.executionKey), expiresAt: resetCandidates[0] || null, config: relayConfig, handleSingleModel: handleSingleModelWrapped, diff --git a/open-sse/services/contextHandoff.ts b/open-sse/services/contextHandoff.ts index ede5f1bb0bb..f6771d74c88 100644 --- a/open-sse/services/contextHandoff.ts +++ b/open-sse/services/contextHandoff.ts @@ -4,13 +4,15 @@ import { type HandoffPayload, upsertHandoff, } from "../../src/lib/db/contextHandoffs.ts"; -import { estimateTokens } from "./contextManager.ts"; +import { estimateTokens, getTokenLimit } from "./contextManager.ts"; import { stripMarkdownCodeFence } from "../utils/aiSdkCompat.ts"; +import { parseModel } from "./model.ts"; export const HANDOFF_WARNING_THRESHOLD = 0.85; export const HANDOFF_EXHAUSTION_THRESHOLD = 0.95; -const MAX_HISTORY_TOKENS_FOR_SUMMARY = 8000; +const FALLBACK_MAX_HISTORY_TOKENS = 8000; +const CONTEXT_SAFETY_MARGIN = 0.4; const DEFAULT_MAX_MESSAGES_FOR_SUMMARY = 30; const DEFAULT_SUMMARY_RESPONSE_TOKENS = 800; const MAX_SUMMARY_LENGTH = 2000; @@ -18,6 +20,7 @@ const MAX_TASK_PROGRESS_LENGTH = 1200; const MAX_DECISIONS = 8; const MAX_ENTITIES = 10; const DEFAULT_TTL_MS = 5 * 60 * 60 * 1000; +const SUMMARIZATION_TIMEOUT_MS = 300000; const OMNI_MODEL_TAG_PATTERN = /(?:\\n|\n|\r)*[^<]+<\/omniModel>(?:\\n|\n|\r)*/g; const inflightHandoffGenerations = new Set(); @@ -122,18 +125,43 @@ function formatMessagesForPrompt(messages: MessageLike[]): string { .join("\n\n"); } -function selectMessagesForSummary(messages: MessageLike[], maxMessages: number): MessageLike[] { +function selectMessagesForSummary( + messages: MessageLike[], + maxMessages: number, + models: string | string[] +): MessageLike[] { + const modelArray = Array.isArray(models) ? models : [models]; + + let minContextLimit = Infinity; + for (const modelStr of modelArray) { + const parsed = parseModel(modelStr); + const provider = parsed.provider || parsed.providerAlias || "unknown"; + const model = parsed.model || modelStr; + const limit = getTokenLimit(provider, model); + minContextLimit = Math.min(minContextLimit, limit); + } + + const maxHistoryTokens = Math.floor(minContextLimit * CONTEXT_SAFETY_MARGIN); + const effectiveLimit = Math.max(maxHistoryTokens, FALLBACK_MAX_HISTORY_TOKENS); + const recentMessages = messages.slice(-maxMessages); let working = [...recentMessages]; while (working.length > 1) { const history = formatMessagesForPrompt(working); - if (estimateTokens(history) <= MAX_HISTORY_TOKENS_FOR_SUMMARY) { + const tokenCount = estimateTokens(history); + if (tokenCount <= effectiveLimit) { return working; } working = working.slice(1); } + if (working.length === 0) { + console.warn( + `[context-handoff] History too large even with single message (${estimateTokens(formatMessagesForPrompt(working))} tokens > ${effectiveLimit} limit for models: ${modelArray.join(", ")})` + ); + } + return working; } @@ -245,6 +273,7 @@ async function generateHandoffAsync(options: { percentUsed: number; messages: MessageLike[]; model: string; + comboTargets?: string[]; expiresAt: string | null; config?: ContextRelayConfig | null; handleSingleModel: (body: Record, modelStr: string) => Promise; @@ -253,9 +282,14 @@ async function generateHandoffAsync(options: { const relayConfig = resolveContextRelayConfig(options.config as Record); const summaryModel = relayConfig.handoffModel || options.model; + + const modelsToConsider = + options.comboTargets && options.comboTargets.length > 0 ? options.comboTargets : [summaryModel]; + const selectedMessages = selectMessagesForSummary( Array.isArray(options.messages) ? options.messages : [], - relayConfig.maxMessagesForSummary + relayConfig.maxMessagesForSummary, + modelsToConsider ); const historyText = formatMessagesForPrompt(selectedMessages); if (!historyText) return; @@ -269,10 +303,56 @@ async function generateHandoffAsync(options: { temperature: 0.1, _omnirouteSkipContextRelay: true, _omnirouteInternalRequest: "context-handoff", + _omnirouteSummarizationTimeout: SUMMARIZATION_TIMEOUT_MS, }; - const response = await options.handleSingleModel(summaryBody, summaryModel); - if (!response.ok) return; + let timeoutId: NodeJS.Timeout | undefined; + const timeoutPromise = new Promise((_, reject) => { + timeoutId = setTimeout( + () => reject(new Error("Summarization timeout")), + SUMMARIZATION_TIMEOUT_MS + ); + }); + + let response: Response; + try { + response = await Promise.race([ + options.handleSingleModel(summaryBody, summaryModel), + timeoutPromise, + ]); + if (timeoutId) clearTimeout(timeoutId); + } catch (err: any) { + if (timeoutId) clearTimeout(timeoutId); + console.warn( + `[context-handoff] Summarization timeout for session ${options.sessionId}: ${err?.message || "unknown"}` + ); + return; + } + + if (!response.ok) { + let errorDetail = `Status ${response.status}`; + try { + const errorText = await response.clone().text(); + if (errorText) { + try { + const errorJson = JSON.parse(errorText); + errorDetail = + errorJson?.error?.message || + errorJson?.error || + errorJson?.message || + errorText.substring(0, 200); + } catch { + errorDetail = errorText.substring(0, 200); + } + } + } catch { + /* ignore */ + } + console.warn( + `[context-handoff] Summarization failed for session ${options.sessionId}: ${errorDetail}` + ); + return; + } let content = ""; try { @@ -287,7 +367,10 @@ async function generateHandoffAsync(options: { } const parsed = parseHandoffJSON(content); - if (!parsed) return; + if (!parsed) { + console.warn(`[context-handoff] Failed to parse handoff JSON for session ${options.sessionId}`); + return; + } upsertHandoff({ sessionId: options.sessionId, @@ -312,6 +395,7 @@ export function maybeGenerateHandoff(options: { percentUsed: number; messages: MessageLike[]; model: string; + comboTargets?: string[]; expiresAt: string | null; config?: ContextRelayConfig | null; handleSingleModel: (body: Record, modelStr: string) => Promise; @@ -331,10 +415,16 @@ export function maybeGenerateHandoff(options: { setImmediate(() => { generateHandoffAsync({ - ...options, sessionId: options.sessionId as string, + comboName: options.comboName, connectionId: options.connectionId as string, + percentUsed: options.percentUsed, + messages: options.messages, + model: options.model, + comboTargets: options.comboTargets, + expiresAt: options.expiresAt, config: relayConfig, + handleSingleModel: options.handleSingleModel, }) .catch((err) => { if (process.env.NODE_ENV !== "test") { diff --git a/open-sse/services/contextManager.ts b/open-sse/services/contextManager.ts index 7aa710b60b5..26764f36cc9 100644 --- a/open-sse/services/contextManager.ts +++ b/open-sse/services/contextManager.ts @@ -6,7 +6,7 @@ */ import { REGISTRY } from "../config/providerRegistry.ts"; -import { getModelContextLimit } from "../../src/lib/modelCapabilities"; +import { getModelContextLimit } from "../../src/lib/modelCapabilities.ts"; // Default token limits per provider (fallbacks when not in registry) const DEFAULT_LIMITS: Record = { @@ -34,6 +34,16 @@ function getEnvOverride(provider: string): number | null { return null; } +// Reserve tokens override from environment variable +function getReserveTokensOverride(): number | null { + const envValue = process.env.CONTEXT_RESERVE_TOKENS; + if (envValue) { + const parsed = parseInt(envValue, 10); + if (!isNaN(parsed) && parsed > 0) return parsed; + } + return null; +} + // Rough chars-per-token ratio for quick estimation const CHARS_PER_TOKEN = 4; @@ -109,7 +119,7 @@ export function compressContext( const provider = options.provider || "default"; const maxTokens = options.maxTokens || getTokenLimit(provider, (body.model as string) || options.model || null); - const reserveTokens = options.reserveTokens || 16000; // Reserve for response + const reserveTokens = options.reserveTokens ?? getReserveTokensOverride() ?? 16000; const targetTokens = maxTokens - reserveTokens; let messages = [...body.messages]; @@ -217,8 +227,8 @@ function compressThinking(messages: Record[]) { // Remove thinking XML tags from string content if (typeof msg.content === "string") { const cleaned = msg.content - .replace(/[\s\S]*?<\/thinking>/g, "") - .replace(/[\s\S]*?<\/antThinking>/g, "") + .replace(/.*?<\/thinking>/gs, "") + .replace(/.*?<\/antThinking>/gs, "") .trim(); return { ...msg, content: cleaned || "[thinking compressed]" }; } diff --git a/open-sse/services/modelFamilyFallback.ts b/open-sse/services/modelFamilyFallback.ts index c75d5bd00bb..b7812ca8779 100644 --- a/open-sse/services/modelFamilyFallback.ts +++ b/open-sse/services/modelFamilyFallback.ts @@ -73,6 +73,7 @@ const MODEL_FAMILIES: Record = { "gemini-2.5-pro-preview-06-05": ["gemini-2.5-pro", "gemini-2.5-pro-exp-03-25"], // Claude Opus family + "claude-opus-4-7": ["claude-opus-4-6", "claude-opus-4-5-20251101", "claude-sonnet-4-6"], "claude-opus-4-6": ["claude-opus-4-6-thinking", "claude-opus-4-5-20251101", "claude-sonnet-4-6"], "claude-opus-4-6-thinking": ["claude-opus-4-6", "claude-opus-4-5-20251101"], diff --git a/open-sse/services/thinkingBudget.ts b/open-sse/services/thinkingBudget.ts index ed000e0e138..ae03ae76db0 100644 --- a/open-sse/services/thinkingBudget.ts +++ b/open-sse/services/thinkingBudget.ts @@ -159,7 +159,8 @@ export function applyThinkingBudget(body, config = null) { // Early exit: strip ALL reasoning/thinking params for models that don't support them. // Sending thinking params to unsupported models (e.g. AG claude-sonnet-4-6) causes 400 errors. const modelStr = typeof body.model === "string" ? body.model : ""; - if (modelStr && !supportsReasoning(modelStr)) { + const isClaude = modelStr.toLowerCase().includes("claude"); + if (modelStr && (!supportsReasoning(modelStr) || (!isClaude && modelStr.includes("gemini")))) { return stripThinkingConfig(body); } diff --git a/open-sse/services/tokenRefresh.ts b/open-sse/services/tokenRefresh.ts index e7a93791e9a..3f71a47f0de 100755 --- a/open-sse/services/tokenRefresh.ts +++ b/open-sse/services/tokenRefresh.ts @@ -579,7 +579,7 @@ export async function refreshKiroToken( /** * Specialized refresh for Qoder OAuth tokens */ -export async function refreshIflowToken(refreshToken, log, proxyConfig = null) { +export async function refreshQoderToken(refreshToken, log, proxyConfig = null) { if (!OAUTH_ENDPOINTS.qoder.token || !PROVIDERS.qoder.clientId || !PROVIDERS.qoder.clientSecret) { log?.warn?.( "TOKEN_REFRESH", @@ -746,7 +746,7 @@ async function _getAccessTokenInternal(provider, credentials, log, proxyConfig = return await refreshQwenToken(credentials.refreshToken, log, proxyConfig); case "qoder": - return await refreshIflowToken(credentials.refreshToken, log, proxyConfig); + return await refreshQoderToken(credentials.refreshToken, log, proxyConfig); case "github": return await refreshGitHubToken(credentials.refreshToken, log, proxyConfig); diff --git a/open-sse/services/usage.ts b/open-sse/services/usage.ts index bfe4fb290b8..d3b8d5123c0 100644 --- a/open-sse/services/usage.ts +++ b/open-sse/services/usage.ts @@ -3,15 +3,24 @@ */ import { PROVIDERS } from "../config/constants.ts"; -import { getAntigravityFetchAvailableModelsUrls } from "../config/antigravityUpstream.ts"; +import { + getAntigravityFetchAvailableModelsUrls, + ANTIGRAVITY_BASE_URLS, +} from "../config/antigravityUpstream.ts"; import { getGlmQuotaUrl } from "../config/glmProvider.ts"; import { safePercentage } from "@/shared/utils/formatting"; import { fetchBailianQuota, type BailianTripleWindowQuota } from "./bailianQuotaFetcher.ts"; import { antigravityUserAgent, + googApiClientHeader, getAntigravityHeaders, getAntigravityLoadCodeAssistMetadata, } from "./antigravityHeaders.ts"; +import { + getAntigravityRemainingCredits, + updateAntigravityRemainingCredits, +} from "../executors/antigravity.ts"; +import { getCreditsMode } from "./antigravityCredits.ts"; // GitHub API config const GITHUB_CONFIG = { @@ -208,7 +217,7 @@ async function getBailianCodingPlanUsage( * @returns {Promise} Usage data with quotas */ export async function getUsageForProvider(connection) { - const { id, provider, accessToken, apiKey, providerSpecificData, projectId } = connection; + const { id, provider, accessToken, apiKey, providerSpecificData, projectId, email } = connection; switch (provider) { case "github": @@ -216,7 +225,7 @@ export async function getUsageForProvider(connection) { case "gemini-cli": return await getGeminiUsage(accessToken, providerSpecificData, projectId); case "antigravity": - return await getAntigravityUsage(accessToken, undefined); + return await getAntigravityUsage(accessToken, providerSpecificData, projectId, id); case "claude": return await getClaudeUsage(accessToken); case "codex": @@ -228,7 +237,7 @@ export async function getUsageForProvider(connection) { case "qwen": return await getQwenUsage(accessToken, providerSpecificData); case "qoder": - return await getIflowUsage(accessToken); + return await getQoderUsage(accessToken); case "glm": case "glmt": return await getGlmUsage(apiKey, providerSpecificData); @@ -881,16 +890,129 @@ function getAntigravityPlanLabel(subscriptionInfo) { return "Free"; } +/** + * Proactive credit balance probe for Antigravity. + * + * Fires a minimal streamGenerateContent request with GOOGLE_ONE_AI credits enabled + * and maxOutputTokens=1 to extract the `remainingCredits` field from the SSE stream. + * This uses ~1 credit but lets us show the balance on the dashboard without waiting + * for a real user request. + * + * Returns the credit balance, or null if the probe failed. + */ +async function probeAntigravityCreditBalance( + accessToken: string, + accountId: string, + projectId?: string | null +): Promise { + try { + if (!projectId) return null; + + // Try all base URLs (some accounts only work with specific endpoints) + for (const baseUrl of ANTIGRAVITY_BASE_URLS) { + const url = `${baseUrl}/v1internal:streamGenerateContent?alt=sse`; + + const sessionId = `-${Math.floor(Math.random() * 9_000_000_000_000_000_000)}`; + const body = { + project: projectId, + model: "gemini-2-flash", + userAgent: "antigravity", + requestType: "agent", + requestId: `credits-probe-${Date.now()}`, + enabledCreditTypes: ["GOOGLE_ONE_AI"], + request: { + model: "gemini-2-flash", + contents: [{ role: "user", parts: [{ text: "hi" }] }], + generationConfig: { maxOutputTokens: 1 }, + sessionId, + }, + }; + + const headers = { + "Content-Type": "application/json", + Authorization: `Bearer ${accessToken}`, + "User-Agent": antigravityUserAgent(), + "X-Goog-Api-Client": googApiClientHeader(), + Accept: "text/event-stream", + }; + + try { + const res = await fetch(url, { + method: "POST", + headers, + body: JSON.stringify(body), + signal: AbortSignal.timeout(10_000), + }); + + if (!res.ok) continue; + + // Read the full SSE response and scan for remainingCredits + const rawSSE = await res.text(); + const lines = rawSSE.split("\n"); + + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (payload === "[DONE]") break; + try { + const parsed = JSON.parse(payload); + if (Array.isArray(parsed?.remainingCredits)) { + const googleCredit = parsed.remainingCredits.find( + (c: { creditType?: string }) => c?.creditType === "GOOGLE_ONE_AI" + ); + if (googleCredit) { + const balance = parseInt(googleCredit.creditAmount, 10); + if (!isNaN(balance)) { + updateAntigravityRemainingCredits(accountId, balance); + return balance; + } + } + } + } catch { + // Skip malformed SSE lines + } + } + } catch { + // Individual endpoint failure; try next + } + } + + return null; + } catch { + // Probe is best-effort — don't let it break the usage fetch + return null; + } +} + /** * Antigravity Usage - Fetch quota from Google Cloud Code API * Uses fetchAvailableModels API which returns ALL models (including Claude) * with per-model quotaInfo (remainingFraction, resetTime). * retrieveUserQuota only returns Gemini models — not suitable for Antigravity. */ -async function getAntigravityUsage(accessToken, providerSpecificData) { +async function getAntigravityUsage( + accessToken, + providerSpecificData, + connectionProjectId?, + connectionId? +) { try { const subscriptionInfo = await getAntigravitySubscriptionInfoCached(accessToken); - const projectId = subscriptionInfo?.cloudaicompanionProject || null; + const projectId = connectionProjectId || subscriptionInfo?.cloudaicompanionProject || null; + + // Derive accountId for credit balance cache. + // Must match executor key: credentials.connectionId + const accountId: string = connectionId || "unknown"; + + // Read cached credit balance (hydrated from DB on first access) + let creditBalance = getAntigravityRemainingCredits(accountId); + + // If no cached balance and credits mode is enabled, fire a minimal probe + const creditsMode = getCreditsMode(); + if (creditBalance === null && creditsMode !== "off") { + creditBalance = await probeAntigravityCreditBalance(accessToken, accountId, projectId); + } // Fetch model list with quota info from fetchAvailableModels let response: Response | null = null; @@ -987,7 +1109,18 @@ async function getAntigravityUsage(accessToken, providerSpecificData) { return { plan: getAntigravityPlanLabel(subscriptionInfo), - quotas, + quotas: { + ...quotas, + ...(creditBalance !== null && { + credits: { + used: 0, + total: 0, + remaining: creditBalance, + unlimited: false, + resetAt: null, + }, + }), + }, subscriptionInfo, }; } catch (error) { @@ -1559,7 +1692,7 @@ async function getQwenUsage(accessToken, providerSpecificData) { /** * Qoder Usage */ -async function getIflowUsage(accessToken) { +async function getQoderUsage(accessToken) { try { // Qoder may have usage endpoint return { message: "Qoder connected. Usage tracked per request." }; diff --git a/open-sse/translator/request/claude-to-gemini.ts b/open-sse/translator/request/claude-to-gemini.ts index 07e65388413..367f12e5539 100644 --- a/open-sse/translator/request/claude-to-gemini.ts +++ b/open-sse/translator/request/claude-to-gemini.ts @@ -25,7 +25,11 @@ export function claudeToGeminiRequest(model, body, stream) { generationConfig: Record; safetySettings: unknown; systemInstruction?: { role: string; parts: Array<{ text: string }> }; - tools?: Array<{ functionDeclarations: Array> }>; + tools?: Array<{ + functionDeclarations?: Array>; + googleSearch?: Record; + googleSearchRetrieval?: Record; + }>; _toolNameMap?: Map; } = { model: model, @@ -152,19 +156,20 @@ export function claudeToGeminiRequest(model, body, stream) { // Map Claude roles to Gemini roles const geminiRole = msg.role === "assistant" ? "model" : "user"; - // Gemini 3+ expects the signature on the first functionCall part in a tool-call + // Gemini 3+ expects the signature on all functionCall parts in a tool-call // batch. If the assistant turn had no explicit thinking block, inject a fallback - // signature into that first functionCall. (#927) + // signature into all functionCalls. if (geminiRole === "model") { const hasFunctionCall = parts.some((p) => p.functionCall); const hasSignature = parts.some((p) => p.thoughtSignature); if (hasFunctionCall && !hasSignature) { - const fcIndex = parts.findIndex((p) => p.functionCall); - if (fcIndex >= 0) { - parts[fcIndex] = { - ...parts[fcIndex], - thoughtSignature: DEFAULT_THINKING_GEMINI_SIGNATURE, - }; + for (let i = 0; i < parts.length; i++) { + if (parts[i].functionCall) { + parts[i] = { + ...parts[i], + thoughtSignature: DEFAULT_THINKING_GEMINI_SIGNATURE, + }; + } } } } diff --git a/open-sse/translator/request/openai-to-gemini.ts b/open-sse/translator/request/openai-to-gemini.ts index e375193109f..26efbe6d12c 100644 --- a/open-sse/translator/request/openai-to-gemini.ts +++ b/open-sse/translator/request/openai-to-gemini.ts @@ -41,6 +41,7 @@ type GeminiGenerationConfig = { }; responseMimeType?: string; responseSchema?: unknown; + stopSequences?: string[] | unknown[]; }; type GeminiFunctionDeclaration = { @@ -213,22 +214,17 @@ function openaiToGeminiBase(model, body, stream, toolNameOptions: GeminiToolName .find((signature) => typeof signature === "string" && signature.length > 0); const shouldUseEmbeddedSignature = !parts.some((p) => p.thoughtSignature); - let embeddedSignatureUsed = false; for (const tc of msg.tool_calls) { if (tc.type !== "function") continue; const args = tryParseJSON(tc.function?.arguments || "{}"); const signatureForToolCall = getGeminiThoughtSignature(tc.id); - const embeddedThoughtSignature = - shouldUseEmbeddedSignature && !embeddedSignatureUsed - ? firstPersistedSignature || - signatureForToolCall || - DEFAULT_THINKING_GEMINI_SIGNATURE - : undefined; - - // Gemini expects the signature on the functionCall part itself. For - // parallel calls, only the first functionCall in the batch carries it. + const embeddedThoughtSignature = shouldUseEmbeddedSignature + ? firstPersistedSignature || signatureForToolCall || DEFAULT_THINKING_GEMINI_SIGNATURE + : undefined; + + // Gemini expects the signature on the functionCall part itself. parts.push({ ...(embeddedThoughtSignature ? { thoughtSignature: embeddedThoughtSignature } : {}), functionCall: { @@ -238,9 +234,6 @@ function openaiToGeminiBase(model, body, stream, toolNameOptions: GeminiToolName }, }); - if (embeddedThoughtSignature) { - embeddedSignatureUsed = true; - } toolCallIds.push(tc.id); } diff --git a/open-sse/utils/proxyFetch.ts b/open-sse/utils/proxyFetch.ts index 36714ea614f..ad1122c6b4f 100644 --- a/open-sse/utils/proxyFetch.ts +++ b/open-sse/utils/proxyFetch.ts @@ -81,7 +81,8 @@ function noProxyMatch(targetUrl) { // Support wildcard matching (e.g. 192.168.* or *.local) if (patternHost.includes("*")) { - const regexStr = "^" + patternHost.replace(/\./g, "\\.").replace(/\*/g, ".*") + "$"; + const regexStr = + "^" + patternHost.replace(/[.*+?^${}()|[\]\\]/g, "\\$&").replace(/\\\*/g, ".*") + "$"; if (new RegExp(regexStr).test(hostname)) return true; } diff --git a/package-lock.json b/package-lock.json index 0f60997a604..28da454ff4a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "omniroute", - "version": "3.6.7", + "version": "3.6.8", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "omniroute", - "version": "3.6.7", + "version": "3.6.8", "hasInstallScript": true, "license": "MIT", "workspaces": [ @@ -51,7 +51,7 @@ "zustand": "^5.0.10" }, "bin": { - "omniroute": "bin/omniroute.ts", + "omniroute": "bin/omniroute.mjs", "omniroute-reset-password": "bin/reset-password.mjs" }, "devDependencies": { @@ -84,7 +84,7 @@ "wtfnode": "^0.10.1" }, "engines": { - "node": ">=20.20.2 <21 || >=22.22.2 <23" + "node": ">=20.20.2 <21 || >=22.22.2 <23 || >=24.0.0 <25" }, "optionalDependencies": { "keytar": "^7.9.0" @@ -20962,7 +20962,7 @@ }, "open-sse": { "name": "@omniroute/open-sse", - "version": "3.6.7" + "version": "3.6.8" } } } diff --git a/package.json b/package.json index 4be879cbc71..c01652b4209 100644 --- a/package.json +++ b/package.json @@ -1,10 +1,10 @@ { "name": "omniroute", - "version": "3.6.7", + "version": "3.6.8", "description": "Smart AI Router with auto fallback — route to FREE & cheap models, zero downtime. Works with Cursor, Cline, Claude Desktop, Codex, and any OpenAI-compatible tool.", "type": "module", "bin": { - "omniroute": "bin/omniroute.ts", + "omniroute": "bin/omniroute.mjs", "omniroute-reset-password": "bin/reset-password.mjs" }, "files": [ @@ -23,6 +23,7 @@ "src/shared/utils/nodeRuntimeSupport.ts", ".env.example", "scripts/postinstall.mjs", + "scripts/postinstallSupport.mjs", "scripts/check-supported-node-runtime.ts", "scripts/sync-env.mjs", "scripts/native-binary-compat.mjs", @@ -34,7 +35,7 @@ "open-sse" ], "engines": { - "node": ">=20.20.2 <21 || >=22.22.2 <23" + "node": ">=20.20.2 <21 || >=22.22.2 <23 || >=24.0.0 <25" }, "keywords": [ "ai", diff --git a/run-responses-test.js b/run-responses-test.js new file mode 100644 index 00000000000..cad4887bbdc --- /dev/null +++ b/run-responses-test.js @@ -0,0 +1,7 @@ +import { test } from "node:test"; +import { execSync } from "child_process"; +console.log("running test..."); +execSync( + "node --import tsx/esm --test tests/integration/chat-pipeline.test.ts --test-name-pattern='chat pipeline serves repeated /v1/responses requests'", + { stdio: "inherit" } +); diff --git a/run_test.js b/run_test.js new file mode 100644 index 00000000000..7c968da9e01 --- /dev/null +++ b/run_test.js @@ -0,0 +1,3 @@ +import fs from "fs"; +import { rotateCallLogs } from "./src/lib/usage/callLogs.js"; +rotateCallLogs(); diff --git a/scripts/pack-artifact-policy.ts b/scripts/pack-artifact-policy.ts index b2530be4dae..8e3711de092 100644 --- a/scripts/pack-artifact-policy.ts +++ b/scripts/pack-artifact-policy.ts @@ -72,6 +72,7 @@ export const PACK_ARTIFACT_ROOT_ALLOWED_EXACT_PATHS: string[] = [ "scripts/check-supported-node-runtime.ts", "scripts/native-binary-compat.mjs", "scripts/postinstall.mjs", + "scripts/postinstallSupport.mjs", "scripts/sync-env.mjs", "src/shared/utils/nodeRuntimeSupport.ts", ]; @@ -89,6 +90,7 @@ export const PACK_ARTIFACT_REQUIRED_PATHS: string[] = [ "package.json", "scripts/native-binary-compat.mjs", "scripts/postinstall.mjs", + "scripts/postinstallSupport.mjs", "src/shared/utils/nodeRuntimeSupport.ts", ]; diff --git a/scripts/postinstall.mjs b/scripts/postinstall.mjs index 1e3f1e67427..87cd21a6594 100644 --- a/scripts/postinstall.mjs +++ b/scripts/postinstall.mjs @@ -22,6 +22,7 @@ import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; import { PUBLISHED_BUILD_ARCH, PUBLISHED_BUILD_PLATFORM } from "./native-binary-compat.mjs"; +import { hasStandaloneAppBundle } from "./postinstallSupport.mjs"; const __filename = fileURLToPath(import.meta.url); const __dirname = dirname(__filename); @@ -174,6 +175,10 @@ async function fixBetterSqliteBinary() { } async function ensureSwcHelpers() { + if (!hasStandaloneAppBundle(ROOT)) { + return; + } + const swcHelpersApp = join(ROOT, "app", "node_modules", "@swc", "helpers"); const swcHelpersRoot = join(ROOT, "node_modules", "@swc", "helpers"); diff --git a/scripts/postinstallSupport.mjs b/scripts/postinstallSupport.mjs new file mode 100644 index 00000000000..bb4f2837f74 --- /dev/null +++ b/scripts/postinstallSupport.mjs @@ -0,0 +1,16 @@ +#!/usr/bin/env node + +import { existsSync } from "node:fs"; +import { join } from "node:path"; + +/** + * Detect whether the current install tree contains the published standalone app bundle. + * Source checkouts should not create `app/` during postinstall because Next.js would + * mis-detect it as a competing App Router root and serve 404s for the real `src/app` routes. + * + * @param {string} rootDir + * @returns {boolean} + */ +export function hasStandaloneAppBundle(rootDir) { + return existsSync(join(rootDir, "app", "server.js")); +} diff --git a/scripts/prepublish.ts b/scripts/prepublish.ts index cc5c6272120..b278a1a81a3 100644 --- a/scripts/prepublish.ts +++ b/scripts/prepublish.ts @@ -372,6 +372,24 @@ if (existsSync(mcpSrcFile)) { } } +// ── Step 8.7: Bundle CLI Entrypoint ────────────────────────── +const cliSrcFile = join(ROOT, "bin", "omniroute.ts"); +const cliDestFile = join(ROOT, "bin", "omniroute.mjs"); + +if (existsSync(cliSrcFile)) { + console.log(" 🔨 Bundling CLI Entrypoint (TypeScript → JavaScript)..."); + try { + execSync( + `npx esbuild bin/omniroute.ts --bundle --platform=node --packages=external --format=esm --outfile=bin/omniroute.mjs`, + { cwd: ROOT, stdio: "inherit" } + ); + execSync(`chmod +x bin/omniroute.mjs`, { cwd: ROOT }); + console.log(" ✅ CLI Entrypoint bundled to bin/omniroute.mjs"); + } catch (err: any) { + console.warn(" ⚠️ CLI bundle error:", err.message); + } +} + // ── Step 9: Copy shared utilities needed at runtime ──────── const sharedApiKey = join(ROOT, "src", "shared", "utils", "apiKey.js"); const sharedApiKeyDest = join(APP_DIR, "src", "shared", "utils"); diff --git a/scratch.mjs b/scripts/scratch.mjs similarity index 100% rename from scratch.mjs rename to scripts/scratch.mjs diff --git a/scripts/scratch/delete_iflow.cjs b/scripts/scratch/delete_iflow.cjs new file mode 100644 index 00000000000..2d80bbf4db9 --- /dev/null +++ b/scripts/scratch/delete_iflow.cjs @@ -0,0 +1,7 @@ +const Database = require('better-sqlite3'); +const db = new Database(process.env.HOME + '/.omniroute/storage.sqlite'); + +console.log("Deleting iflow connections..."); +const stmt = db.prepare("DELETE FROM provider_connections WHERE provider = 'iflow'"); +const info = stmt.run(); +console.log(`Deleted ${info.changes} rows.`); diff --git a/scripts/scratch/dump-auth-groups.ts b/scripts/scratch/dump-auth-groups.ts new file mode 100644 index 00000000000..f598461d4f1 --- /dev/null +++ b/scripts/scratch/dump-auth-groups.ts @@ -0,0 +1,23 @@ +import { + APIKEY_PROVIDERS, + SEARCH_PROVIDERS, + AUDIO_ONLY_PROVIDERS, + WEB_COOKIE_PROVIDERS, +} from "./src/shared/constants/providers"; + +console.log("SEARCH_PROVIDERS", !!SEARCH_PROVIDERS, Object.keys(SEARCH_PROVIDERS || {})); +console.log("AUDIO_ONLY", !!AUDIO_ONLY_PROVIDERS); +console.log("WEB_COOKIE", !!WEB_COOKIE_PROVIDERS); + +// Determine auth type group for a provider id +function getAuthGroup(providerId: string) { + if (WEB_COOKIE_PROVIDERS && WEB_COOKIE_PROVIDERS[providerId]) return "web-cookie"; + if (SEARCH_PROVIDERS && SEARCH_PROVIDERS[providerId]) return "search"; + if (AUDIO_ONLY_PROVIDERS && AUDIO_ONLY_PROVIDERS[providerId]) return "audio"; + if (APIKEY_PROVIDERS && APIKEY_PROVIDERS[providerId]) return "apikey"; + return "apikey"; +} + +console.log("grok-web returns:", getAuthGroup("grok-web")); +console.log("perplexity-search returns:", getAuthGroup("perplexity-search")); +console.log("openai returns:", getAuthGroup("openai")); diff --git a/scripts/scratch/fix-tests.js b/scripts/scratch/fix-tests.js new file mode 100644 index 00000000000..f6ac99e1ffd --- /dev/null +++ b/scripts/scratch/fix-tests.js @@ -0,0 +1,30 @@ +const fs = require("fs"); + +function addAnyCast(filePath) { + let content = fs.readFileSync(filePath, "utf8"); + // Match "const varname = await func({...});" and make it "const varname: any = await func({...});" + content = content.replace( + /(const\s+\w+)\s*=\s*(await\s+(?:usageService|usageFetcher)\.getUsageForProvider\()/g, + "$1: any = $2" + ); + fs.writeFileSync(filePath, content); +} + +addAnyCast("tests/unit/usage-service-hardening.test.ts"); +addAnyCast("tests/unit/usage-fetcher-antigravity.test.ts"); + +let bailian = fs.readFileSync("tests/unit/bailian-quota-fetcher.test.ts", "utf8"); +// Fix missing window properties in test typing +bailian = bailian.replace(/const quota = /g, "const quota: any = "); +fs.writeFileSync("tests/unit/bailian-quota-fetcher.test.ts", bailian); + +let routeTest = fs.readFileSync("tests/unit/token-refresh-route-service.test.ts", "utf8"); +// Fix provider mocks typing +routeTest = routeTest.replace(/github: \{/g, '"github": {'); // Fix github +routeTest = routeTest.replace( + /(refreshWithRetry|log\.entries).*?(toBe|equal).*?;/g, + (match) => match +); // just ignore +fs.writeFileSync("tests/unit/token-refresh-route-service.test.ts", routeTest); + +console.log("Fixes applied."); diff --git a/scripts/scratch/overlap.js b/scripts/scratch/overlap.js new file mode 100644 index 00000000000..93ac1df151d --- /dev/null +++ b/scripts/scratch/overlap.js @@ -0,0 +1,21 @@ +import { + APIKEY_PROVIDERS, + SEARCH_PROVIDERS, + AUDIO_ONLY_PROVIDERS, + WEB_COOKIE_PROVIDERS, +} from "./src/shared/constants/providers.ts"; + +const apiKeys = Object.keys(APIKEY_PROVIDERS); +console.log("Searching overlap in APIKEY_PROVIDERS:"); +console.log( + "Search overlap:", + Object.keys(SEARCH_PROVIDERS).filter((k) => apiKeys.includes(k)) +); +console.log( + "Audio overlap:", + Object.keys(AUDIO_ONLY_PROVIDERS).filter((k) => apiKeys.includes(k)) +); +console.log( + "Web Cookie overlap:", + Object.keys(WEB_COOKIE_PROVIDERS).filter((k) => apiKeys.includes(k)) +); diff --git a/scripts/scratch/overlap.ts b/scripts/scratch/overlap.ts new file mode 100644 index 00000000000..ceb10a0aadc --- /dev/null +++ b/scripts/scratch/overlap.ts @@ -0,0 +1,8 @@ +import { APIKEY_PROVIDERS, SEARCH_PROVIDERS } from "./src/shared/constants/providers.ts"; + +const apiKeys = Object.keys(APIKEY_PROVIDERS); +console.log( + "Overlap:", + Object.keys(SEARCH_PROVIDERS).filter((k) => apiKeys.includes(k)) +); +console.log("Is perplexity-search in APIKEY?", "perplexity-search" in APIKEY_PROVIDERS); diff --git a/scripts/scratch/query_db.cjs b/scripts/scratch/query_db.cjs new file mode 100644 index 00000000000..3bc7a75efd8 --- /dev/null +++ b/scripts/scratch/query_db.cjs @@ -0,0 +1,18 @@ +const Database = require('better-sqlite3'); +const db = new Database(process.env.HOME + '/.omniroute/storage.sqlite'); + +const providers = db.prepare("SELECT * FROM provider_connections").all(); +console.log("=== provider_connections ==="); +console.log(providers.filter(p => JSON.stringify(p).toLowerCase().includes('iflow'))); + +const combos = db.prepare("SELECT * FROM combos").all(); +console.log("=== combos ==="); +console.log(combos.filter(c => JSON.stringify(c).toLowerCase().includes('iflow'))); + +const settings = db.prepare("SELECT * FROM settings").all(); +console.log("=== settings ==="); +console.log(settings.filter(s => JSON.stringify(s).toLowerCase().includes('iflow'))); + +const apiKeys = db.prepare("SELECT * FROM api_keys").all(); +console.log("=== api_keys ==="); +console.log(apiKeys.filter(k => JSON.stringify(k).toLowerCase().includes('iflow'))); diff --git a/scripts/scratch/query_db.js b/scripts/scratch/query_db.js new file mode 100644 index 00000000000..7f94c56e1c9 --- /dev/null +++ b/scripts/scratch/query_db.js @@ -0,0 +1,10 @@ +const Database = require("better-sqlite3"); +const db = new Database(process.env.HOME + "/.omniroute/storage.sqlite"); +console.log("=== provider_connections containing iflow ==="); +console.log( + db + .prepare( + "SELECT id, provider_id, alias, account_id FROM provider_connections WHERE provider_id LIKE '%iflow%' OR alias LIKE '%iflow%' OR account_id LIKE '%iflow%'" + ) + .all() +); diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 696ffb11944..324afdf2523 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -121,19 +121,19 @@ const STRATEGY_GUIDANCE_FALLBACK = { example: "Example: Multiple accounts of the same model to distribute usage evenly.", }, auto: { - when: "Use when you have one preferred model and only want fallback on failure.", - avoid: "Avoid when you need balanced load between models.", - example: "Example: Primary coding model with cheaper backup for outages.", + when: "Use when you want multi-factor scoring based on cost, latency, and quality.", + avoid: "Avoid when you need strict priority ordering or historical persistence.", + example: "Example: Balance requests between models with different strengths.", }, lkgp: { - when: "Use when you have one preferred model and only want fallback on failure.", - avoid: "Avoid when you need balanced load between models.", - example: "Example: Primary coding model with cheaper backup for outages.", + when: "Use when you want routing based on historical success rates and performance.", + avoid: "Avoid when historical data is limited or unreliable.", + example: "Example: Route to models with proven track records for specific tasks.", }, "context-optimized": { - when: "Use when you have one preferred model and only want fallback on failure.", - avoid: "Avoid when you need balanced load between models.", - example: "Example: Primary coding model with cheaper backup for outages.", + when: "Use when you need to optimize for context window usage across models.", + avoid: "Avoid when models have similar context lengths or simple tasks.", + example: "Example: Distribute long conversations across models with large context windows.", }, }; @@ -240,30 +240,30 @@ const STRATEGY_RECOMMENDATIONS_FALLBACK = { ], }, auto: { - title: "Fail-safe baseline", - description: "Use one primary model and keep fallback chain short and reliable.", + title: "Multi-factor optimization", + description: "Routes based on real-time scoring of cost, latency, quality, and health.", tips: [ - "Put your most reliable model first.", - "Keep 1-2 backup models with similar quality.", - "Use safe retries to absorb transient provider failures.", + "Let the engine balance across multiple factors automatically.", + "Monitor which factors drive routing decisions in the logs.", + "Use for complex workloads where no single factor dominates.", ], }, lkgp: { - title: "Fail-safe baseline", - description: "Use one primary model and keep fallback chain short and reliable.", + title: "History-based routing", + description: "Routes based on historical success rates and persistent performance data.", tips: [ - "Put your most reliable model first.", - "Keep 1-2 backup models with similar quality.", - "Use safe retries to absorb transient provider failures.", + "Let success history accumulate before relying on this strategy.", + "Models with better track records get preference over time.", + "Ideal for stable workloads with consistent model availability.", ], }, "context-optimized": { - title: "Fail-safe baseline", - description: "Use one primary model and keep fallback chain short and reliable.", + title: "Context-aware distribution", + description: "Routes to optimize context window usage and conversation continuity.", tips: [ - "Put your most reliable model first.", - "Keep 1-2 backup models with similar quality.", - "Use safe retries to absorb transient provider failures.", + "Best for long conversations that span multiple requests.", + "Selects models with appropriate context capacity automatically.", + "Use when context limits are a bottleneck for your workload.", ], }, }; @@ -314,7 +314,7 @@ const COMBO_TEMPLATE_FALLBACK = { balancedDesc: "Least-used routing to spread demand over time.", freeStackTitle: "Free Stack ($0)", freeStackDesc: - "Round-robin across all free providers: Kiro, iFlow, Qwen, Gemini CLI. Zero cost, never stops.", + "Round-robin across all free providers: Kiro, Qoder, Qwen, Gemini CLI. Zero cost, never stops.", paidPremiumTitle: "Paid Premium", paidPremiumDesc: "Round-robin across paid subscriptions: Cursor, Antigravity. Top-tier models, distributed load.", @@ -1812,7 +1812,7 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) { const [builderConnectionId, setBuilderConnectionId] = useState(COMBO_BUILDER_AUTO_CONNECTION); const [builderComboRefName, setBuilderComboRefName] = useState(""); const [builderError, setBuilderError] = useState(""); - const [builderStage, setBuilderStage] = useState(COMBO_BUILDER_STAGES[0]); + const [builderStage, setBuilderStage] = useState(COMBO_BUILDER_STAGES[0]); const [showAdvanced, setShowAdvanced] = useState(false); const [config, setConfig] = useState(combo?.config || {}); const [showStrategyNudge, setShowStrategyNudge] = useState(false); @@ -2724,7 +2724,7 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) { t={t} config={config} activeProviders={activeProviders} - onChange={(nextIntelligentConfig) => + onChange={(nextIntelligentConfig: any) => setConfig((previousConfig) => ({ ...previousConfig, ...nextIntelligentConfig, diff --git a/src/app/(dashboard)/dashboard/playground/page.tsx b/src/app/(dashboard)/dashboard/playground/page.tsx index 91906d705a9..3b36cf8e8e7 100644 --- a/src/app/(dashboard)/dashboard/playground/page.tsx +++ b/src/app/(dashboard)/dashboard/playground/page.tsx @@ -1,6 +1,6 @@ "use client"; -import { useState, useEffect, useCallback, useRef } from "react"; +import { useState, useEffect, useCallback, useRef, useMemo } from "react"; import { useTranslations } from "next-intl"; import { Card, Button, Select, Badge } from "@/shared/components"; import { ALIAS_TO_ID } from "@/shared/constants/providers"; @@ -185,18 +185,21 @@ export default function PlaygroundPage() { const t = useTranslations("playground"); // Get translated endpoint options - const getEndpointOptions = () => [ - { value: "chat", label: t("endpointOptions.chat") }, - { value: "responses", label: t("endpointOptions.responses") }, - { value: "images", label: t("endpointOptions.images") }, - { value: "embeddings", label: t("endpointOptions.embeddings") }, - { value: "speech", label: t("endpointOptions.speech") }, - { value: "transcription", label: t("endpointOptions.transcription") }, - { value: "video", label: t("endpointOptions.video") }, - { value: "music", label: t("endpointOptions.music") }, - { value: "rerank", label: t("endpointOptions.rerank") }, - { value: "search", label: t("endpointOptions.search") }, - ]; + const endpointOptions = useMemo( + () => [ + { value: "chat", label: t("endpointOptions.chat") }, + { value: "responses", label: t("endpointOptions.responses") }, + { value: "images", label: t("endpointOptions.images") }, + { value: "embeddings", label: t("endpointOptions.embeddings") }, + { value: "speech", label: t("endpointOptions.speech") }, + { value: "transcription", label: t("endpointOptions.transcription") }, + { value: "video", label: t("endpointOptions.video") }, + { value: "music", label: t("endpointOptions.music") }, + { value: "rerank", label: t("endpointOptions.rerank") }, + { value: "search", label: t("endpointOptions.search") }, + ], + [t] + ); const [models, setModels] = useState([]); const [providers, setProviders] = useState([]); @@ -495,7 +498,7 @@ export default function PlaygroundPage() {

{t("pricingRatesFormat")}

- ", "") - .replace("", ""), - }} - /> + {t.rich("ratesDescription", { + strong: (chunks) => {chunks}, + })}

diff --git a/src/shared/components/ProxyConfigModal.tsx b/src/shared/components/ProxyConfigModal.tsx index 00a2351d304..54fd53630d7 100644 --- a/src/shared/components/ProxyConfigModal.tsx +++ b/src/shared/components/ProxyConfigModal.tsx @@ -354,7 +354,10 @@ export default function ProxyConfigModal({ const title = level === "global" ? t("titleGlobal") - : `${t(`level${level.charAt(0).toUpperCase() + level.slice(1)}` as any)} Proxy — ${levelLabel || levelId || ""}`; + : t("titleLevel", { + level: t(`level${level.charAt(0).toUpperCase() + level.slice(1)}` as any), + label: levelLabel || levelId || "", + }); return ( diff --git a/src/shared/components/RequestLoggerV2.tsx b/src/shared/components/RequestLoggerV2.tsx index 1899b5f7726..48f0c7e7e47 100644 --- a/src/shared/components/RequestLoggerV2.tsx +++ b/src/shared/components/RequestLoggerV2.tsx @@ -116,27 +116,35 @@ export default function RequestLoggerV2() { const t = useTranslations("requestLogger"); // Get translated status filters - const getStatusFilters = () => [ - { key: "all", label: t("statusFilters.all"), icon: "" }, - { key: "error", label: t("statusFilters.error"), icon: "error" }, - { key: "ok", label: t("statusFilters.success"), icon: "check_circle" }, - { key: "combo", label: t("statusFilters.combo"), icon: "hub" }, - ]; + const statusFilters = useMemo( + () => [ + { key: "all", label: t("statusFilters.all"), icon: "" }, + { key: "error", label: t("statusFilters.error"), icon: "error" }, + { key: "ok", label: t("statusFilters.success"), icon: "check_circle" }, + { key: "combo", label: t("statusFilters.combo"), icon: "hub" }, + ], + [t] + ); // Get translated columns - const getColumns = () => [ - { key: "status", label: t("columns.status") }, - { key: "model", label: t("columns.model") }, - { key: "requestedModel", label: t("columns.requested") }, - { key: "provider", label: t("columns.provider") }, - { key: "protocol", label: t("columns.protocol") }, - { key: "account", label: t("columns.account") }, - { key: "apiKey", label: t("columns.apiKey") }, - { key: "combo", label: t("columns.combo") }, - { key: "tokens", label: t("columns.tokens") }, - { key: "duration", label: t("columns.duration") }, - { key: "time", label: t("columns.time") }, - ]; + const columns = useMemo( + () => [ + { key: "status", label: t("columns.status") }, + { key: "cacheSource", label: t("columns.cacheSource") }, + { key: "model", label: t("columns.model") }, + { key: "requestedModel", label: t("columns.requested") }, + { key: "provider", label: t("columns.provider") }, + { key: "protocol", label: t("columns.protocol") }, + { key: "account", label: t("columns.account") }, + { key: "apiKey", label: t("columns.apiKey") }, + { key: "combo", label: t("columns.combo") }, + { key: "tokens", label: t("columns.tokens") }, + { key: "tps", label: t("columns.tps") }, + { key: "duration", label: t("columns.duration") }, + { key: "time", label: t("columns.time") }, + ], + [t] + ); const [logs, setLogs] = useState([]); const [loading, setLoading] = useState(true); @@ -159,7 +167,7 @@ export default function RequestLoggerV2() { // Column visibility with localStorage persistence const [visibleColumns, setVisibleColumns] = useState(() => { - const defaultVisible = Object.fromEntries(getColumns().map((c) => [c.key, true])); + const defaultVisible = Object.fromEntries(columns.map((c) => [c.key, true])); if (typeof window === "undefined") return defaultVisible; try { const saved = localStorage.getItem("loggerVisibleColumns"); @@ -525,7 +533,7 @@ export default function RequestLoggerV2() { {/* Quick Filters */}
{/* Status Filters */} - {getStatusFilters().map((f) => ( + {statusFilters.map((f) => (