diff --git a/.dockerignore b/.dockerignore
index 9f7dea5a316..069c1e67812 100644
--- a/.dockerignore
+++ b/.dockerignore
@@ -67,3 +67,42 @@ images
clipr
omnirouteCloud
omnirouteSite
+
+# Temporary/Scratch Folders
+_*
+
+# CI/CD and Version Control (that are not actual code)
+.github
+.husky
+.omc
+
+# Test Configs and Reports
+playwright.config.ts
+vitest*.ts
+audit-report.json
+sonar-project.properties
+
+# Deployment Configs
+docker-compose*.yml
+fly.toml
+
+# Consistent with .gitignore
+.DS_Store
+.idea/
+.config/
+.data/
+.omnivscodeagent/
+*.sqlite-*
+*.tsbuildinfo
+next-env.d.ts
+security-analysis/
+.analysis/
+antigravity-manager-analysis/
+.sisyphus/
+.plans/
+app.__qa_backup/
+.app-build-backup-*/
+.gitnexus
+.worktrees
+.next-playwright/
+cloud/
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 6778e667fa7..f8b3ff80817 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -15,7 +15,8 @@ permissions:
contents: read
env:
- CI_NODE_VERSION: "22.22.2"
+ CI_NODE_VERSION: "24"
+ CI_NODE_24_VERSION: "24"
jobs:
lint:
@@ -185,6 +186,26 @@ jobs:
- run: npm run check:node-runtime
- run: npm run test:unit
+ node-24-compat:
+ name: Node 24 Compatibility
+ runs-on: ubuntu-latest
+ timeout-minutes: 15
+ needs: build
+ env:
+ JWT_SECRET: ci-test-secret-with-sufficient-length-for-validation
+ API_KEY_SECRET: ci-test-api-key-secret-long
+ DISABLE_SQLITE_AUTO_BACKUP: "true"
+ steps:
+ - uses: actions/checkout@v6
+ - uses: actions/setup-node@v6
+ with:
+ node-version: ${{ env.CI_NODE_24_VERSION }}
+ cache: npm
+ - run: npm ci
+ - run: npm run check:node-runtime
+ - run: npm run build
+ - run: npm run test:unit
+
test-coverage:
name: Coverage
runs-on: ubuntu-latest
@@ -413,6 +434,7 @@ jobs:
- build
- package-artifact
- test-unit
+ - node-24-compat
- test-coverage
- sonarqube
- coverage-pr-comment
diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml
index c8bdb4b061e..84974b88a43 100644
--- a/.github/workflows/electron-release.yml
+++ b/.github/workflows/electron-release.yml
@@ -71,13 +71,11 @@ jobs:
deb_ext: .deb
steps:
- - name: Checkout
- uses: actions/checkout@v6
-
- - name: Setup Node.js
+ - uses: actions/checkout@v6
+ - name: Setup Node
uses: actions/setup-node@v6
with:
- node-version: 22
+ node-version: 24
cache: npm
- name: Cache node_modules
diff --git a/.github/workflows/npm-publish.yml b/.github/workflows/npm-publish.yml
index bb56d71bc7c..3f38d745708 100644
--- a/.github/workflows/npm-publish.yml
+++ b/.github/workflows/npm-publish.yml
@@ -38,7 +38,7 @@ permissions:
packages: write
env:
- NPM_PUBLISH_NODE_VERSION: "22.22.2"
+ NPM_PUBLISH_NODE_VERSION: "24"
jobs:
publish:
diff --git a/.gitignore b/.gitignore
index c4abb01ea6f..783740a7d79 100644
--- a/.gitignore
+++ b/.gitignore
@@ -170,3 +170,9 @@ docs/superpowers/
# GitNexus local index
.gitnexus
.worktrees
+bin/omniroute.mjs
+
+# Consistent with .dockerignore / .npmignore
+.omc/
+audit-report.json
+bun.lock
diff --git a/.node-version b/.node-version
index 2bd5a0a98a3..a45fd52cc58 100644
--- a/.node-version
+++ b/.node-version
@@ -1 +1 @@
-22
+24
diff --git a/.npmignore b/.npmignore
index e0a5b886543..cc14c145c20 100644
--- a/.npmignore
+++ b/.npmignore
@@ -76,3 +76,27 @@ app/_*/
app/coverage/
app/logs/
app/tests/
+
+# Consistent with .gitignore and .dockerignore
+.DS_Store
+.idea/
+.config/
+.data/
+.omnivscodeagent/
+.omc/
+*.sqlite-*
+*.tsbuildinfo
+security-analysis/
+.analysis/
+antigravity-manager-analysis/
+.sisyphus/
+.plans/
+app.__qa_backup/
+.app-build-backup-*/
+.gitnexus
+.worktrees
+.next-playwright/
+test-results/
+playwright-report/
+blob-report/
+coverage/
diff --git a/.nvmrc b/.nvmrc
index 2bd5a0a98a3..a45fd52cc58 100644
--- a/.nvmrc
+++ b/.nvmrc
@@ -1 +1 @@
-22
+24
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 6791a35c2d5..b4f2a1d0630 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,14 +4,33 @@
---
-## [3.6.7] — 2026-04-16
+## [3.6.8] — 2026-04-16
### ✨ New Features
+- **feat(core):** Add full support for Node.js 24 LTS (Krypton) environments with continuous integration coverage (#1340)
+- **feat(dashboard):** Display Antigravity credit balance in dashboard Limits & Quotas (#1338)
- **feat(i18n):** Add internationalization support for combo features and dashboard components; sync translations across 31 keys (#1318)
-
-### 🐛 Bug Fixes
-
+- **feat(providers):** Add Claude Opus 4.7 to Claude Code OAuth models natively with extended context and caching (#1347)
+- **feat(core):** Add stopSequences support and expand tool definitions to include Google Search capabilities
+- **security:** Resolve GitHub CodeQL scan alerts and enforce deep SSRF mitigations
+
+### 🐛 Bug Fixes
+
+- **fix(db):** Prevent native module ABI load crashes from assuming database corruption and skipping databases
+- **fix(db):** Increase mass-migration threshold from 5 to 50 pending migrations to protect legacy users upgrading node
+- **fix(db):** Prevent migration runner safety aborts from triggering on fresh `DATA_DIR` installations by detecting new databases (#1328)
+- **fix(mcp):** Checkpoint and close MCP audit SQLite database safely on process signals and shutdown (#1348)
+- **fix(mcp):** Fully decouple MCP audit SQLite connection caching via globalThis to fix unhandled teardown in standalone Next.js chunks (#1349)
+- **fix(cli):** Avoid creating app router directory during postinstall initialization on non-built source trees (#1351)
+- **fix(codex):** Correctly translate `system` role to `developer` in input array to unlock GPT-5 automatic prompt caching (#1346)
+- **fix(core):** Pass client headers to executor in chatCore (#1335)
+- **fix(providers):** Separate test batch calls and ignore unknown connections
+- **fix(providers):** Add grok-web SSO cookie validation handler (#1334)
+- **fix(db):** Preserve key_value settings (dashboard passwords, saved aliases) across DB heuristic recreation cycles (#1333)
+- **fix(routing):** Allow combo fallback to cascade context overflow 400 errors instead of immediate aborts (#1331)
+- **fix(core):** Resolve thinking leaks, consecutive roles, and missing thoughtSignatures for Antigravity translator (#1316)
+- **fix(providers):** Default to batch testing execution blocks for web, search, and audio modalities to prevent connection timeouts
- **fix(cli):** Resolve Node 22 TS entrypoint incompatibility by using esbuild compilation (#1315)
- **fix(chat):** Preserve max_output_tokens for Responses API targets in chatCore sanitization (#1313)
- **fix(api):** API Manager usage stats showing 0 for all registered keys (#1310)
diff --git a/Dockerfile b/Dockerfile
index 1aafbfd65fb..706ba212c1d 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -1,4 +1,4 @@
-FROM node:22.22.2-trixie-slim AS builder
+FROM node:24.14.1-trixie-slim AS builder
WORKDIR /app
RUN apt-get update \
@@ -13,7 +13,7 @@ RUN if [ -f package-lock.json ]; then npm ci --no-audit --no-fund; else npm inst
COPY . ./
RUN mkdir -p /app/data && npm run build -- --webpack
-FROM node:22.22.2-trixie-slim AS runner-base
+FROM node:24.14.1-trixie-slim AS runner-base
WORKDIR /app
LABEL org.opencontainers.image.title="omniroute" \
diff --git a/GEMINI.md b/GEMINI.md
new file mode 100644
index 00000000000..ead803af35a
--- /dev/null
+++ b/GEMINI.md
@@ -0,0 +1,15 @@
+# Security and Cleanliness Rules for AI Assistants
+
+## 1. File Placement & Organization
+
+- **Test Files**: ALL unit tests, integration tests, ecosystem tests, or Vitest files MUST strictly be placed within the `tests/` directory (e.g., `tests/unit/`, `tests/integration/`). NEVER create test files in the project root (`/`).
+- **Scripts and Utilities**: ALL maintenance, debugging, generation, or experimental scripts (`.cjs`, `.mjs`, `.js`, `.ts`) MUST be placed strictly inside the `scripts/` directory or `scripts/scratch/` for temporary one-offs. NEVER dump loose scripts in the project root (`/`).
+
+**The Project Root MUST ONLY CONTAIN:**
+
+- Configuration files (`vitest.config.ts`, `next.config.mjs`, `eslint.config.mjs`, etc.)
+- Dependency files (`package.json`, `package-lock.json`)
+- Documentation files (`README.md`, `CHANGELOG.md`, `AGENTS.md`)
+- CI/CD files and ignore definitions (`.gitignore`, `.dockerignore`)
+
+When creating _any_ validation tests or one-off logic scripts, default to using `scripts/scratch/` or the `tests/unit/` directories according to your goals. Do not pollute the `/` root context.
diff --git a/README.md b/README.md
index 72360615163..7f45c559c04 100644
--- a/README.md
+++ b/README.md
@@ -695,7 +695,7 @@ During deep debugging, long histories with tool results quickly exceed provider
```txt
Combo: "maximize-claude"
- 1. cc/claude-opus-4-6
+ 1. cc/claude-opus-4-7
2. glm/glm-4.7
3. if/kimi-k2-thinking
@@ -719,7 +719,7 @@ Outcome: stable free coding workflow
```txt
Combo: "always-on"
- 1. cc/claude-opus-4-6
+ 1. cc/claude-opus-4-7
2. cx/gpt-5.2-codex
3. glm/glm-4.7
4. minimax/MiniMax-M2.1
@@ -1511,7 +1511,7 @@ OmniRoute v3.6 is built as an operational platform, not just a relay proxy.
```txt
Combo: "my-coding-stack"
- 1. cc/claude-opus-4-6
+ 1. cc/claude-opus-4-7
2. nvidia/llama-3.3-70b
3. glm/glm-4.7
4. if/kimi-k2-thinking
@@ -1650,7 +1650,7 @@ Dashboard → Providers → Connect Claude Code
→ 5-hour + weekly quota tracking
Models:
- cc/claude-opus-4-6
+ cc/claude-opus-4-7
cc/claude-sonnet-4-5-20250929
cc/claude-haiku-4-5-20251001
```
@@ -1850,7 +1850,7 @@ Dashboard → Combos → Create New
Name: premium-coding
Models:
- 1. cc/claude-opus-4-6 (Subscription primary)
+ 1. cc/claude-opus-4-7 (Subscription primary)
2. glm/glm-4.7 (Cheap backup, $0.6/1M)
3. minimax/MiniMax-M2.1 (Cheapest fallback, $0.20/1M)
@@ -1880,7 +1880,7 @@ Cost: $0 forever!
Settings → Models → Advanced:
OpenAI API Base URL: http://localhost:20128/v1
OpenAI API Key: [from OmniRoute dashboard]
- Model: cc/claude-opus-4-6
+ Model: cc/claude-opus-4-7
```
### Claude Code
@@ -1990,7 +1990,7 @@ opencode
**Rate limiting**
- Subscription quota out → Fallback to GLM/MiniMax
-- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Add combo: `cc/claude-opus-4-7 → glm/glm-4.7 → if/kimi-k2-thinking`
**OAuth token expired**
@@ -2322,9 +2322,23 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 📊 Star History
-## Stargazers over time
-
-## [](https://starchart.cc/diegosouzapw/OmniRoute)
+
+
+
+
+
+
+
+
+## 🌍 StarMapper
+
+
+
+
+
+
+
+
## 🙏 Acknowledgments
diff --git a/docker-compose.yml b/docker-compose.yml
index 4eae86ba200..7e24dbce844 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -20,6 +20,7 @@
x-common: &common
restart: unless-stopped
+ stop_grace_period: 40s
env_file: .env
environment:
- DATA_DIR=/app/data # Must match the volume mount below
@@ -101,7 +102,7 @@ services:
# Adjust paths below to match YOUR host system.
- ~/.local/bin:/host-local/bin:ro
# Node global binaries (adjust node version path)
- # - ~/.nvm/versions/node/v22.16.0/bin:/host-node/bin:ro
+ # - ~/.nvm/versions/node/v24.14.1/bin:/host-node/bin:ro
# ── Host config mounts (read-write) ──
- ~/.codex:/host-home/.codex:rw
- ~/.claude:/host-home/.claude:rw
diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md
index 8a9c13fa62f..f9a69755bbe 100644
--- a/docs/TROUBLESHOOTING.md
+++ b/docs/TROUBLESHOOTING.md
@@ -15,7 +15,7 @@ Common problems and solutions for OmniRoute.
| No logs written to disk | Set `APP_LOG_TO_FILE=true` and verify call log capture is enabled |
| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
-| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Login crash / blank page | Check Node.js version — see [Node.js Compatibility](#nodejs-compatibility) below |
| `dlopen` / `slice is not valid mach-o file` (macOS) | Run `cd $(npm root -g)/omniroute/app && npm rebuild better-sqlite3 && omniroute` — see [macOS native module rebuild](#macos-native-module-rebuild) below |
| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
@@ -27,10 +27,7 @@ Common problems and solutions for OmniRoute.
### Login page crashes or shows "Module self-registration" error
-**Cause:** You are running a Node.js version outside OmniRoute's approved secure runtime floor. Two cases matter:
-
-1. **Node.js 24+**: `better-sqlite3` is not supported here and startup can fail hard.
-2. **Older Node 20/22 patch levels**: the runtime may start, but it falls below the patched security floor OmniRoute now requires.
+**Cause:** You are running a Node.js version outside OmniRoute's approved secure runtime floor. The most common case is running an older Node 20, 22, or 24 patch level that falls below the patched security floor OmniRoute requires.
**Symptoms:**
@@ -40,16 +37,16 @@ Common problems and solutions for OmniRoute.
**Fix:**
-1. Install a patched Node.js 22 LTS release (recommended):
+1. Install a supported Node.js LTS release (recommended: Node.js 24.x):
```bash
- nvm install 22.22.2
- nvm use 22.22.2
+ nvm install 24
+ nvm use 24
```
-2. Verify your version: `node --version` should show `v22.22.2` or newer on the 22.x LTS line
+2. Verify your version: `node --version` should show `v24.0.0` or newer on the 24.x LTS line
3. Reinstall OmniRoute: `npm install -g omniroute`
4. Restart: `omniroute`
-> **Supported secure versions:** `>=20.20.2 <21` or `>=22.22.2 <23`. Node.js 24+ is **not supported**.
+> **Supported secure versions:** `>=20.20.2 <21`, `>=22.22.2 <23`, or `>=24.0.0 <25`. Node.js 24.x LTS (Krypton) is fully supported.
### macOS: `dlopen` / "slice is not valid mach-o file"
@@ -64,7 +61,7 @@ Common problems and solutions for OmniRoute.
- Full example:
```
-dlopen(/Users//.nvm/versions/node/v24.13.1/lib/node_modules/omniroute/app/node_modules/better-sqlite3/build/Release/better_sqlite3.node, 0x0001): tried: '...' (slice is not valid mach-o file)
+dlopen(/Users//.nvm/versions/node/v24.14.1/lib/node_modules/omniroute/app/node_modules/better-sqlite3/build/Release/better_sqlite3.node, 0x0001): tried: '...' (slice is not valid mach-o file)
```
**Fix — rebuild for your local environment (no Node.js downgrade required):**
@@ -75,7 +72,7 @@ npm rebuild better-sqlite3
omniroute
```
-> **Note:** This recompiles the native binding against your local Node.js version and CPU architecture, resolving the binary mismatch. The officially supported secure range is now **`>=20.20.2 <21` or `>=22.22.2 <23`** (`engines` field in `package.json`). If you are on Node.js 24, the rebuild may silence this specific startup error but other issues can still occur — moving to a patched Node.js 22 LTS release remains the recommended path.
+> **Note:** This recompiles the native binding against your local Node.js version and CPU architecture, resolving the binary mismatch. The officially supported range is **`>=20.20.2 <21`, `>=22.22.2 <23`, or `>=24.0.0 <25`** (`engines` field in `package.json`). Node.js 24.x LTS (Krypton) is fully supported with `better-sqlite3` v12.x.
---
diff --git a/docs/USER_GUIDE.md b/docs/USER_GUIDE.md
index b4892aabb38..a639755c1d6 100644
--- a/docs/USER_GUIDE.md
+++ b/docs/USER_GUIDE.md
@@ -55,7 +55,7 @@ Complete guide for configuring providers, creating combos, integrating CLI tools
```
Combo: "maximize-claude"
- 1. cc/claude-opus-4-6 (use subscription fully)
+ 1. cc/claude-opus-4-7 (use subscription fully)
2. glm/glm-4.7 (cheap backup when quota out)
3. if/kimi-k2-thinking (free emergency fallback)
@@ -83,7 +83,7 @@ Quality: Production-ready models
```
Combo: "always-on"
- 1. cc/claude-opus-4-6 (best quality)
+ 1. cc/claude-opus-4-7 (best quality)
2. cx/gpt-5.2-codex (second subscription)
3. glm/glm-4.7 (cheap, resets daily)
4. minimax/MiniMax-M2.1 (cheapest, 5h reset)
@@ -121,7 +121,7 @@ Dashboard → Providers → Connect Claude Code
→ 5-hour + weekly quota tracking
Models:
- cc/claude-opus-4-6
+ cc/claude-opus-4-7
cc/claude-sonnet-4-5-20250929
cc/claude-haiku-4-5-20251001
```
@@ -230,7 +230,7 @@ Dashboard → Combos → Create New
Name: premium-coding
Models:
- 1. cc/claude-opus-4-6 (Subscription primary)
+ 1. cc/claude-opus-4-7 (Subscription primary)
2. glm/glm-4.7 (Cheap backup, $0.6/1M)
3. minimax/MiniMax-M2.1 (Cheapest fallback, $0.20/1M)
@@ -259,7 +259,7 @@ Cost: $0 forever!
Settings → Models → Advanced:
OpenAI API Base URL: http://localhost:20128/v1
OpenAI API Key: [from omniroute dashboard]
- Model: cc/claude-opus-4-6
+ Model: cc/claude-opus-4-7
```
### Claude Code
@@ -313,7 +313,7 @@ Edit `~/.openclaw/openclaw.json`:
Provider: OpenAI Compatible
Base URL: http://localhost:20128/v1
API Key: [from dashboard]
-Model: cc/claude-opus-4-6
+Model: cc/claude-opus-4-7
```
---
@@ -552,7 +552,7 @@ For the full environment variable reference, see the [README](../README.md).
View all available models
-**Claude Code (`cc/`)** — Pro/Max: `cc/claude-opus-4-6`, `cc/claude-sonnet-4-5-20250929`, `cc/claude-haiku-4-5-20251001`
+**Claude Code (`cc/`)** — Pro/Max: `cc/claude-opus-4-7`, `cc/claude-sonnet-4-5-20250929`, `cc/claude-haiku-4-5-20251001`
**Codex (`cx/`)** — Plus/Pro: `cx/gpt-5.2-codex`, `cx/gpt-5.1-codex-max`
@@ -745,7 +745,7 @@ Define global fallback chains that apply across all requests:
```
Chain: production-fallback
- 1. cc/claude-opus-4-6
+ 1. cc/claude-opus-4-7
2. gh/gpt-5.1-codex
3. glm/glm-4.7
```
diff --git a/docs/openapi.yaml b/docs/openapi.yaml
index d608323f5a0..93e916306c5 100644
--- a/docs/openapi.yaml
+++ b/docs/openapi.yaml
@@ -1,7 +1,7 @@
openapi: 3.1.0
info:
title: OmniRoute API
- version: 3.6.7
+ version: 3.6.8
description: |
OmniRoute is a local-first AI API proxy router. It provides an OpenAI-compatible
endpoint that routes requests to multiple AI providers with load balancing,
diff --git a/electron/package.json b/electron/package.json
index 6ee0649e069..f37885dae66 100644
--- a/electron/package.json
+++ b/electron/package.json
@@ -1,6 +1,6 @@
{
"name": "omniroute-desktop",
- "version": "3.6.7",
+ "version": "3.6.8",
"description": "OmniRoute Desktop Application",
"main": "main.js",
"author": {
diff --git a/llm.txt b/llm.txt
index 8df769de43f..b14bae387e0 100644
--- a/llm.txt
+++ b/llm.txt
@@ -8,7 +8,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo
**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
-**Current version:** 3.6.6
+**Current version:** 3.6.8
## Tech Stack
@@ -279,7 +279,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo
└── .env.example # Environment variable template
```
-## Key Features (v3.6.6)
+## Key Features (v3.6.8)
### Core Proxy
- **60+ AI providers** with automatic format translation
diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts
index 7ddaa5009c8..02369d2eb5f 100644
--- a/open-sse/config/providerRegistry.ts
+++ b/open-sse/config/providerRegistry.ts
@@ -277,6 +277,7 @@ export const REGISTRY: Record = {
tokenUrl: "https://console.anthropic.com/v1/oauth/token",
},
models: [
+ { id: "claude-opus-4-7", name: "Claude Opus 4.7" },
{ id: "claude-opus-4-6", name: "Claude Opus 4.6" },
{ id: "claude-sonnet-4-6", name: "Claude 4.6 Sonnet" },
{ id: "claude-opus-4-5-20251101", name: "Claude 4.5 Opus" },
diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts
index d56403b98d1..83eeabe720c 100644
--- a/open-sse/executors/antigravity.ts
+++ b/open-sse/executors/antigravity.ts
@@ -7,8 +7,11 @@ import { classify429, decide429, type Decision } from "../services/antigravity42
import {
injectCreditsField,
shouldRetryWithCredits,
+ shouldUseCreditsFirst,
+ getCreditsMode,
handleCreditsFailure,
} from "../services/antigravityCredits.ts";
+import { persistCreditBalance, getAllPersistedCreditBalances } from "@/lib/db/creditBalance";
import { obfuscateSensitiveWords } from "../services/antigravityObfuscation.ts";
const MAX_RETRY_AFTER_MS = 60_000;
@@ -28,11 +31,30 @@ const creditsExhaustedUntil = new Map();
* Per-account GOOGLE_ONE_AI remaining credit balance cache.
* Populated from the final SSE chunk's `remainingCredits` field after every
* successful credit-injected request. Keyed by accountId.
+ * On first access, hydrated from the DB-persisted balances so values survive restarts.
*/
const creditBalanceCache = new Map();
+let creditCacheHydrated = false;
+
+function hydrateCreditCacheFromDb(): void {
+ if (creditCacheHydrated) return;
+ creditCacheHydrated = true;
+ try {
+ const persisted = getAllPersistedCreditBalances();
+ for (const [accountId, balance] of persisted) {
+ // Only fill in accounts not already populated by a live SSE response
+ if (!creditBalanceCache.has(accountId)) {
+ creditBalanceCache.set(accountId, balance);
+ }
+ }
+ } catch {
+ // DB not ready yet (build phase, etc.) — ignore silently
+ }
+}
/** Read the last-known GOOGLE_ONE_AI credit balance for a given account. */
export function getAntigravityRemainingCredits(accountId: string): number | null {
+ hydrateCreditCacheFromDb();
const balance = creditBalanceCache.get(accountId);
return balance !== undefined ? balance : null;
}
@@ -40,6 +62,12 @@ export function getAntigravityRemainingCredits(accountId: string): number | null
/** Update the balance cache — called when we parse `remainingCredits` from an SSE stream. */
export function updateAntigravityRemainingCredits(accountId: string, balance: number): void {
creditBalanceCache.set(accountId, balance);
+ // Persist to DB so the value survives server restarts
+ try {
+ persistCreditBalance(accountId, balance);
+ } catch {
+ // Non-critical — in-memory cache is the primary source
+ }
}
function isCreditsExhausted(accountId: string): boolean {
@@ -150,13 +178,27 @@ export class AntigravityExecutor extends BaseExecutor {
// Antigravity rejects synthetic thought text, but Gemini 3+ requires any
// returned thoughtSignature metadata to survive model tool-call turns.
const parts =
- c.parts?.filter((p) => !p.thought && (hasFunctionCall || !p.thoughtSignature)) || [];
+ c.parts?.filter((p) => {
+ // Drop empty text parts
+ if (typeof p.text === "string" && p.text === "") return false;
+ // Drop empty functionCalls
+ if (p.functionCall && !p.functionCall.name) return false;
+
+ return !p.thought && (hasFunctionCall || !p.thoughtSignature);
+ }) || [];
return { ...c, role, parts };
}) || [];
- const contents = normalizedContents.filter((c) =>
- Array.isArray(c.parts) ? c.parts.length > 0 : true
- );
+ // Merge consecutive same-role entries and filter out empty sequences
+ const contents = [];
+ for (const c of normalizedContents) {
+ if (!Array.isArray(c.parts) || c.parts.length === 0) continue;
+ if (contents.length > 0 && contents[contents.length - 1].role === c.role) {
+ contents[contents.length - 1].parts.push(...c.parts);
+ } else {
+ contents.push(c);
+ }
+ }
const transformedRequest = {
...body.request,
@@ -412,17 +454,31 @@ export class AntigravityExecutor extends BaseExecutor {
// non-streaming Response so chatCore's non-streaming path stays unchanged.
const upstreamStream = true;
- // Account ID for credits-exhausted tracking.
- // Key must match getAntigravityUsage() in fetcher.ts (providerSpecificData?.email || sub).
- // credentials.email and credentials.sub are populated from the same OAuth token store,
- // so the cache keys written here and read in the fetcher will always match.
- const accountId: string = credentials?.email || credentials?.sub || "unknown";
+ // Account ID for credits tracking.
+ // Use connectionId as the stable cache key — it's available in both the executor
+ // (via credentials.connectionId) and the usage fetcher (via connection.id).
+ // The email-based key was unreliable because email isn't always on the credentials object.
+ const accountId: string = credentials?.connectionId || "unknown";
+
+ // Resolve credits mode once per execute() call. "always" injects
+ // enabledCreditTypes: ["GOOGLE_ONE_AI"] on the first request so the
+ // preflight normal call is skipped entirely.
+ const creditsMode = getCreditsMode();
+ const useCreditsFirst = shouldUseCreditsFirst(credentials?.accessToken || "", creditsMode);
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
const url = this.buildUrl(model, upstreamStream, urlIndex);
const headers = this.buildHeaders(credentials, upstreamStream);
mergeUpstreamExtraHeaders(headers, upstreamExtraHeaders);
- const transformedBody = await this.transformRequest(model, body, upstreamStream, credentials);
+ let transformedBody = await this.transformRequest(model, body, upstreamStream, credentials);
+
+ // Credits-first: inject GOOGLE_ONE_AI upfront so we never try the normal
+ // quota path. If credits are exhausted / disabled shouldUseCreditsFirst()
+ // returns false and we fall back to the legacy retry-on-429 flow.
+ if (useCreditsFirst) {
+ transformedBody = injectCreditsField(transformedBody);
+ log?.debug?.("AG_CREDITS", "Credits-first enabled (ANTIGRAVITY_CREDITS=always)");
+ }
// Initialize retry counter for this URL
if (!retryAttemptsByUrl[urlIndex]) {
@@ -457,17 +513,29 @@ export class AntigravityExecutor extends BaseExecutor {
// 1. Try to parse explicit retry time from message
const parsedRetryMs = this.parseRetryFromErrorMessage(errorMessage);
- // 2. Classify 429
- const category = classify429(errorMessage);
+ // 2. Classify 429 (pass header-parsed retry hint as fallback
+ // signal — multi-hour Retry-After upgrades rate_limited to
+ // quota_exhausted so the GOOGLE_ONE_AI credits retry fires).
+ const effectiveRetryHintMs = retryMs ?? parsedRetryMs ?? null;
+ const category = classify429(errorMessage, effectiveRetryHintMs);
// 3. For quota_exhausted, attempt Google One AI credits retry FIRST!
+ // Skip if credits were already injected on the first call
+ // (creditsMode === "always") — no point re-running with the
+ // same body. Record the failure so the 5h breaker kicks in.
+ const creditsAlreadyInjected =
+ (transformedBody as { enabledCreditTypes?: unknown }).enabledCreditTypes != null;
+
+ if (category === "quota_exhausted" && creditsAlreadyInjected) {
+ handleCreditsFailure(credentials?.accessToken || "");
+ log?.warn?.("AG_CREDITS", "Credits-first request 429'd — credits likely exhausted");
+ markCreditsExhausted(accountId);
+ }
+
if (
category === "quota_exhausted" &&
- shouldRetryWithCredits(
- credentials?.accessToken || "",
- process.env.ANTIGRAVITY_CREDITS === "1" ||
- process.env.ANTIGRAVITY_CREDITS === "true"
- )
+ !creditsAlreadyInjected &&
+ shouldRetryWithCredits(credentials?.accessToken || "", creditsMode !== "off")
) {
log?.info?.("AG_CREDITS", "Retrying with Google One AI credits");
const creditsBody = injectCreditsField(transformedBody);
@@ -613,7 +681,7 @@ export class AntigravityExecutor extends BaseExecutor {
// For non-streaming clients, collect the SSE stream and return a synthetic
// non-streaming Response so chatCore doesn't need to handle SSE conversion.
if (!stream) {
- return this.collectStreamToResponse(
+ const collected = await this.collectStreamToResponse(
response,
model,
url,
@@ -622,6 +690,82 @@ export class AntigravityExecutor extends BaseExecutor {
log,
signal
);
+ // When credits were injected (credits-first or credits-retry), the
+ // synthetic body contains _remainingCredits — mirror it into the
+ // balance cache so the dashboard stays fresh.
+ try {
+ const syntheticJson = await collected.response.clone().json();
+ const rc = syntheticJson?._remainingCredits;
+ if (Array.isArray(rc)) {
+ const googleCredit = rc.find(
+ (c: { creditType?: string }) => c?.creditType === "GOOGLE_ONE_AI"
+ );
+ if (googleCredit) {
+ const balance = parseInt(googleCredit.creditAmount, 10);
+ if (!isNaN(balance)) updateAntigravityRemainingCredits(accountId, balance);
+ }
+ }
+ } catch {
+ /* balance cache is best-effort */
+ }
+ return collected;
+ }
+
+ // Streaming path: wrap the response body in a pass-through TransformStream
+ // that extracts remainingCredits from the final SSE chunk(s) without
+ // consuming the stream. The client receives the unmodified SSE data.
+ if (response.body) {
+ let sseBuffer = "";
+ const passThrough = new TransformStream({
+ transform(chunk, controller) {
+ controller.enqueue(chunk);
+ // Accumulate text to scan for remainingCredits
+ try {
+ const text = new TextDecoder().decode(chunk, { stream: true });
+ sseBuffer += text;
+ } catch {
+ /* decoding best-effort */
+ }
+ },
+ flush() {
+ // Parse the accumulated SSE data for remainingCredits
+ try {
+ const lines = sseBuffer.split("\n");
+ for (const line of lines) {
+ const trimmed = line.trim();
+ if (!trimmed.startsWith("data:")) continue;
+ const payload = trimmed.slice(5).trim();
+ if (!payload || payload === "[DONE]") continue;
+ try {
+ const parsed = JSON.parse(payload);
+ if (Array.isArray(parsed?.remainingCredits)) {
+ const googleCredit = parsed.remainingCredits.find(
+ (c) => c?.creditType === "GOOGLE_ONE_AI"
+ );
+ if (googleCredit) {
+ const balance = parseInt(googleCredit.creditAmount, 10);
+ if (!isNaN(balance)) {
+ updateAntigravityRemainingCredits(accountId, balance);
+ }
+ }
+ }
+ } catch {
+ /* skip malformed lines */
+ }
+ }
+ } catch {
+ /* credits extraction is best-effort */
+ }
+ sseBuffer = "";
+ },
+ });
+ const tappedBody = response.body.pipeThrough(passThrough);
+ const tappedResponse = new Response(tappedBody, {
+ status: response.status,
+ statusText: response.statusText,
+ headers: response.headers,
+ });
+ return { response: tappedResponse, url, headers, transformedBody };
}
return { response, url, headers, transformedBody };
diff --git a/open-sse/executors/base.ts b/open-sse/executors/base.ts
index ee953afe10f..c623802d7df 100644
--- a/open-sse/executors/base.ts
+++ b/open-sse/executors/base.ts
@@ -390,6 +390,7 @@ export class BaseExecutor {
// Only supported for specific Claude models per Anthropic docs
if (extendedContext) {
const EXTENDED_CONTEXT_MODELS = [
+ "claude-opus-4-7",
"claude-opus-4-6",
"claude-sonnet-4-6",
"claude-sonnet-4-5",
diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts
index 9a631bcfda7..e04f7d172f5 100644
--- a/open-sse/executors/codex.ts
+++ b/open-sse/executors/codex.ts
@@ -224,6 +224,39 @@ function hoistSystemMessagesToInstructions(body: Record): void
body.input = filteredInput;
}
+/**
+ * Convert role=system messages in `input` to role=developer.
+ *
+ * GPT-5 models support the `developer` role in input, but reject `system`.
+ * Unlike hoistSystemMessagesToInstructions(), this keeps the content inside
+ * the `input` array where it benefits from OpenAI's automatic prompt caching.
+ *
+ * OpenAI's prompt caching matches on the serialized prefix of the `input` array
+ * (+ tools). The `instructions` field is NOT included in the cache key for
+ * GPT-5 models. Moving system prompts from `input` to `instructions` therefore
+ * removes them from the cacheable prefix, resulting in 0% cache hit rates.
+ *
+ * Ref: https://community.openai.com/t/caching-is-borked-for-gpt-5-models/1359574
+ * Ref: https://community.openai.com/t/no-caching-with-model-responses/1338627
+ */
+function convertSystemToDeveloperRole(body: Record): void {
+ if (!Array.isArray(body.input)) return;
+
+ for (const itemValue of body.input) {
+ if (!itemValue || typeof itemValue !== "object" || Array.isArray(itemValue)) {
+ continue;
+ }
+
+ const item = itemValue as Record;
+ const role = typeof item.role === "string" ? item.role : "";
+ const type = typeof item.type === "string" ? item.type : "";
+ const isSystemMessage = role === "system" && (!type || type === "message");
+ if (isSystemMessage) {
+ item.role = "developer";
+ }
+ }
+}
+
function normalizeCodexTools(body: Record): void {
if (!Array.isArray(body.tools)) return;
@@ -436,11 +469,47 @@ export class CodexExecutor extends BaseExecutor {
body.service_tier = requestDefaults.serviceTier;
}
- // If no instructions provided, inject default Codex instructions
- // NOTE: must run before the passthrough return — Codex upstream rejects
- // requests without instructions even when the body is forwarded as-is.
- if (!body.instructions || body.instructions.trim() === "") {
- body.instructions = CODEX_DEFAULT_INSTRUCTIONS;
+ // ── System prompt handling: cache-aware strategy ──
+ //
+ // For GPT-5 models, OpenAI's automatic prompt caching only considers the
+ // `input` array content (+ tools). The `instructions` field is NOT included
+ // in the cache prefix computation. Moving system prompts from `input` into
+ // `instructions` therefore removes them from the cacheable prefix, causing
+ // 0% cache hit rates even with identical repeated requests.
+ //
+ // For native passthrough (client sends Responses API format directly):
+ // - Convert system → developer role in-place (Codex accepts developer but rejects system)
+ // - Only inject minimal instructions if the field is completely empty
+ // - Do NOT inject CODEX_DEFAULT_INSTRUCTIONS (it would bloat the non-cached field)
+ //
+ // For translated requests (from Chat Completions format):
+ // - Continue hoisting system messages to instructions (legacy behavior)
+ // - Inject CODEX_DEFAULT_INSTRUCTIONS as fallback
+ //
+ // Ref: https://community.openai.com/t/caching-is-borked-for-gpt-5-models/1359574
+ // Ref: https://community.openai.com/t/no-caching-with-model-responses/1338627
+ if (nativeCodexPassthrough) {
+ // Passthrough path: keep system prompts in input for caching.
+ // Convert system → developer role since Codex rejects role=system in input.
+ convertSystemToDeveloperRole(body);
+
+ // Codex still requires a non-empty instructions field.
+ // Use a minimal placeholder if the client didn't provide one.
+ if (
+ !body.instructions ||
+ (typeof body.instructions === "string" && body.instructions.trim() === "")
+ ) {
+ body.instructions = "Follow the developer instructions in the conversation.";
+ }
+ } else {
+ // Translated path: hoist system messages to instructions (legacy behavior).
+ if (
+ !body.instructions ||
+ (typeof body.instructions === "string" && body.instructions.trim() === "")
+ ) {
+ body.instructions = CODEX_DEFAULT_INSTRUCTIONS;
+ }
+ hoistSystemMessagesToInstructions(body);
}
if (!storeEnabled) {
@@ -449,10 +518,6 @@ export class CodexExecutor extends BaseExecutor {
body.store = responsesStoreMarker;
}
- // Cursor can send native Responses payloads with role=system items inside `input`.
- // Codex rejects system messages there; they must be folded into `instructions`.
- hoistSystemMessagesToInstructions(body);
-
// Codex Responses only supports function tools with non-empty names.
// Cursor may include custom tools (e.g. ApplyPatch) that work locally but are
// invalid upstream, and translation bugs can leave orphaned/empty tool_choice names.
diff --git a/open-sse/executors/cursor.ts b/open-sse/executors/cursor.ts
index c3faf2b2905..e914d6a0182 100644
--- a/open-sse/executors/cursor.ts
+++ b/open-sse/executors/cursor.ts
@@ -1,4 +1,5 @@
declare const EdgeRuntime: string | undefined;
+import crypto from "node:crypto";
/**
* CursorExecutor — Handles communication with the Cursor IDE API.
*
@@ -30,7 +31,6 @@ import {
import { estimateUsage } from "../utils/usageTracking.ts";
import { getCursorVersion } from "../utils/cursorVersionDetector.ts";
import { FORMATS } from "../translator/formats.ts";
-import crypto from "crypto";
import { v5 as uuidv5 } from "uuid";
import zlib from "zlib";
diff --git a/open-sse/executors/perplexity-web.ts b/open-sse/executors/perplexity-web.ts
index b0232dd5c23..6b1c287d7d5 100644
--- a/open-sse/executors/perplexity-web.ts
+++ b/open-sse/executors/perplexity-web.ts
@@ -6,6 +6,7 @@
* completions format and Perplexity's internal protocol.
*/
+import crypto from "node:crypto";
import { BaseExecutor, type ExecuteInput } from "./base.ts";
const PPLX_SSE_ENDPOINT = "https://www.perplexity.ai/rest/sse/perplexity_ask";
@@ -33,9 +34,9 @@ const CITATION_RE = /\[\d+\]/g;
const GROK_TAG_RE = /]*>.*?<\/grok:[^>]*>/gs;
const GROK_SELF_RE = /]*\/>/g;
const XML_DECL_RE = /<[?]xml[^?]*[?]>/g;
-const SCRIPT_RE = /