Skip to content
Merged
27 changes: 24 additions & 3 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -15,11 +15,27 @@
- **feat(i18n):** Add internationalization support for combo features and dashboard components; sync translations across 31 keys (#1318)
- **feat(providers):** Add Claude Opus 4.7 to Claude Code OAuth models natively with extended context and caching (#1347)
- **feat(core):** Add stopSequences support and expand tool definitions to include Google Search capabilities
- **security:** Resolve GitHub CodeQL scan alerts and enforce deep SSRF mitigations
- **feat(auth):** Enforce dashboard session authentication on all management API routes, preventing unauthenticated access to configuration endpoints
- **feat(runtime):** Add hot-reloadable guardrails and model diagnostics for real-time rule evaluation without restarts
- **feat(core):** Add payload rules, tag-based routing, and scheduled budget systems for fine-grained request governance
- **feat(providers):** Expose Antigravity preview model aliases and Gemini CLI onboarding flow for first-time setup
- **feat(antigravity):** Add client model aliases and thoughtSignature bypass modes for Antigravity OAuth connections
- **feat(providers):** Expand image provider registry with extended model support including SD3.5, FLUX, and DALL-E 3 HD configurations
- **feat(combos):** Add new routing strategies and full i18n support for agent features section across 31 languages

### 🔒 Security

- **security:** Resolve 18 GitHub CodeQL scan alerts including ReDoS, incomplete sanitization, and bad HTML filtering regexp patterns
- **fix(auth):** Seal privilege escalation vector by enforcing JWT session checking exclusively on `/api/keys` management endpoints (#1353)
- **fix(providers):** Resolve Codex token refresh race condition via mutex `getAccessToken` preventing `refresh_token_reused` Auth0 revocations

### 🔧 Maintenance & Architecture

- **refactor(core):** Split CLI runner and decouple migration engine for extensibility (#1358)
- **refactor(audit):** Rewire audit dashboard from dead in-memory `configAudit` store to live SQLite `audit_log` table — 331+ hidden compliance entries now visible in `/dashboard/audit`
- **build(deps):** Bump `softprops/action-gh-release` from v2 to v3
- **ci:** Bump GitHub Actions CI node-version to Node.js 24 natively
- **fix(types):** Resolve TypeScript compilation errors in `claudeCodeCompatible.ts` (type predicates, `cache_control` index access) and `proxyFetch.ts` (`signal` nullability)

### 🐛 Bug Fixes

Expand All @@ -29,20 +45,25 @@
- **fix(mcp):** Checkpoint and close MCP audit SQLite database safely on process signals and shutdown (#1348)
- **fix(mcp):** Fully decouple MCP audit SQLite connection caching via globalThis to fix unhandled teardown in standalone Next.js chunks (#1349)
- **fix(cli):** Avoid creating app router directory during postinstall initialization on non-built source trees (#1351)
- **fix(providers):** Resolve token refresh race condition causing Codex accounts to be erroneously flagged with `refresh_token_reused` by Auth0
- **fix(auth):** Seal privilege escalation vector by enforcing JWT session checking exclusively on `/api/keys` management endpoints (#1353)
- **fix(codex):** Correctly translate `system` role to `developer` in input array to unlock GPT-5 automatic prompt caching (#1346)
- **fix(core):** Pass client headers to executor in chatCore (#1335)
- **fix(providers):** Separate test batch calls and ignore unknown connections
- **fix(providers):** Add grok-web SSO cookie validation handler (#1334)
- **fix(db):** Preserve key_value settings (dashboard passwords, saved aliases) across DB heuristic recreation cycles (#1333)
- **fix(routing):** Allow combo fallback to cascade context overflow 400 errors instead of immediate aborts (#1331)
- **fix(core):** Resolve thinking leaks, consecutive roles, and missing thoughtSignatures for Antigravity translator (#1316)
- **fix(translator):** Only apply thoughtSignature to the first `functionCall` part in Gemini parallel tool calls, preventing duplicate signatures
- **fix(providers):** Default to batch testing execution blocks for web, search, and audio modalities to prevent connection timeouts
- **fix(cli):** Resolve Node 22 TS entrypoint incompatibility by using esbuild compilation (#1315)
- **fix(chat):** Preserve max_output_tokens for Responses API targets in chatCore sanitization (#1313)
- **fix(api):** API Manager usage stats showing 0 for all registered keys (#1310)
- **fix(api):** Support image-only models in catalog and allow authless search providers to bypass validation requirements
- **fix(routes):** Require prompts for media generation requests (`/images`, `/videos`, `/music`), returning 400 on missing payloads
- **fix(dashboard):** Auto-scroll ActivityHeatmap to show current date (#1309)
- **fix(dashboard):** Restore horizontal layout with `w-max` wrapper in heatmap components
- **fix(i18n):** Update `nodeIncompatibleHint` to recommend Node 24 LTS across all 31 languages
- **fix(i18n):** Add Chinese i18n support to remaining dashboard components (`Loading.tsx`, `DataTable`, etc.)
- **fix(requestLogger):** Add missing `cacheSource` and `tps` columns to i18n log detail views

## [3.6.6] — 2026-04-15

Expand Down
16 changes: 9 additions & 7 deletions bin/nodeRuntimeSupport.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -3,11 +3,13 @@
export const SECURE_NODE_LINES = Object.freeze([
Object.freeze({ major: 20, minor: 20, patch: 2 }),
Object.freeze({ major: 22, minor: 22, patch: 2 }),
Object.freeze({ major: 24, minor: 0, patch: 0 }),
]);

export const RECOMMENDED_NODE_VERSION = "22.22.2";
export const SUPPORTED_NODE_RANGE = ">=20.20.2 <21 || >=22.22.2 <23";
export const SUPPORTED_NODE_DISPLAY = "Node.js 20.20.2+ (20.x LTS) or 22.22.2+ (22.x LTS)";
export const RECOMMENDED_NODE_VERSION = "24.14.1";
export const SUPPORTED_NODE_RANGE = ">=20.20.2 <21 || >=22.22.2 <23 || >=24.0.0 <25";
export const SUPPORTED_NODE_DISPLAY =
"Node.js 20.20.2+ (20.x LTS), 22.22.2+ (22.x LTS), or 24.0.0+ (24.x LTS)";

function formatVersion(version) {
return `${version.major}.${version.minor}.${version.patch}`;
Expand Down Expand Up @@ -50,8 +52,8 @@ export function getNodeRuntimeSupport(version = process.versions.node) {
reason = "supported";
} else if (secureFloor) {
reason = "below-security-floor";
} else if (parsed.major >= 24) {
reason = "native-addon-incompatible";
} else if (parsed.major >= 25) {
reason = "unreleased-major";
}

return {
Expand All @@ -73,8 +75,8 @@ export function getNodeRuntimeWarning(version = process.versions.node) {
return `Node.js ${support.nodeVersion} is below the patched minimum ${support.minimumSecureVersion} for this LTS line.`;
}

if (support.reason === "native-addon-incompatible") {
return `Node.js ${support.nodeVersion} is outside the supported LTS lines and may fail at runtime because better-sqlite3 does not support Node.js 24+ here.`;
if (support.reason === "unreleased-major") {
return `Node.js ${support.nodeVersion} is outside the supported LTS lines. OmniRoute currently supports Node.js 20.x, 22.x, and 24.x.`;
}

return `Node.js ${support.nodeVersion} is outside OmniRoute's approved secure runtime policy.`;
Expand Down
28 changes: 28 additions & 0 deletions open-sse/config/imageRegistry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -493,3 +493,31 @@ export function getAllImageModels() {
export function getImageModelAliases() {
return IMAGE_MODEL_ALIASES;
}

export function getImageModelEntry(modelStr) {
if (!modelStr) return null;

const alias = IMAGE_MODEL_ALIASES[modelStr];
if (alias) {
const modelConfig = findImageModelConfig(alias.provider, alias.model);
return {
provider: alias.provider,
model: alias.model,
inputModalities: alias.inputModalities || modelConfig?.inputModalities || ["text"],
description: alias.description || modelConfig?.description || undefined,
};
}

const { provider, model } = parseImageModel(modelStr);
if (!provider || !model) return null;

const modelConfig = findImageModelConfig(provider, model);
if (!modelConfig) return null;

return {
provider,
model,
inputModalities: modelConfig.inputModalities || ["text"],
description: modelConfig.description || undefined,
};
}
4 changes: 0 additions & 4 deletions open-sse/executors/perplexity-web.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,8 +33,6 @@ const CITATION_RE = /\[\d+\]/g;
const GROK_TAG_RE = /<grok:[^>]*>.*?<\/grok:[^>]*>/gs;
const GROK_SELF_RE = /<grok:[^>]*\/>/g;
const XML_DECL_RE = /<[?]xml[^?]*[?]>/g;
const SCRIPT_RE = /<script\b[^>]*>.*?<\/script>/gis;
const SCRIPT_TAG_RE = /<\/?script\b[^>]*>/gi;
const RESPONSE_TAG_RE = /<\/?response\b[^>]*>/gi;
const MULTI_SPACE = / {2,}/g;
const MULTI_NL = /\n{3,}/g;
Expand Down Expand Up @@ -109,8 +107,6 @@ function cleanResponse(text: string, strip = true): string {
t = t.replace(GROK_TAG_RE, "");
t = t.replace(GROK_SELF_RE, "");
t = t.replace(RESPONSE_TAG_RE, "");
t = t.replace(SCRIPT_RE, ""); // lgtm[js/incomplete-multi-character-sanitization]
t = t.replace(SCRIPT_TAG_RE, ""); // lgtm[js/incomplete-multi-character-sanitization]
if (strip) {
t = t.replace(MULTI_SPACE, " ");
t = t.replace(MULTI_NL, "\n\n");
Expand Down
19 changes: 18 additions & 1 deletion open-sse/handlers/chatCore.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1452,12 +1452,20 @@ export async function handleChatCore({
}
}

// ── Proactive Context Compression (Phase 4) ──
// Check if context exceeds 85% of limit and compress proactively before sending to provider.
// This prevents "prompt too long" errors for large-but-not-full contexts.
if (translatedBody && translatedBody.messages && Array.isArray(translatedBody.messages)) {
const estimatedTokens = estimateTokens(JSON.stringify(translatedBody.messages));
const contextLimit = getTokenLimit(provider, effectiveModel);
const COMPRESSION_THRESHOLD = 0.85;
const threshold = Math.floor(contextLimit * COMPRESSION_THRESHOLD);

log?.debug?.(
"CONTEXT",
`Checking compression: ${estimatedTokens} tokens vs ${threshold} threshold (${contextLimit} limit)`
);

if (estimatedTokens > threshold) {
log?.info?.(
"CONTEXT",
Expand All @@ -1468,6 +1476,7 @@ export async function handleChatCore({
provider,
model: effectiveModel,
maxTokens: contextLimit,
reserveTokens: 0,

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

Setting reserveTokens to 0 here effectively disables the proactive compression for contexts that fall between the 85% threshold and the 100% limit.

In compressContext, the target token count is calculated as maxTokens - reserveTokens. If reserveTokens is 0, the target is the full contextLimit. This means that if a request is triggered at 90% of the limit, compressContext will see that it already fits within the 100% target and will not perform any compression.

Furthermore, passing 0 here overrides the environment variable CONTEXT_RESERVE_TOKENS and the default value of 16000 defined in contextManager.ts. To allow the service to use its configured default or environment override, you should avoid passing 0 explicitly.

Suggested change
reserveTokens: 0,
reserveTokens: undefined,

});

if (compressionResult.compressed) {
Expand Down Expand Up @@ -1495,8 +1504,15 @@ export async function handleChatCore({
layers: "layers" in stats ? stats.layers : undefined,
},
});
} else {
log?.debug?.("CONTEXT", `Compression not applied: context already fits within target`);
}
}
} else {
log?.debug?.(
"CONTEXT",
`Skipping compression check: translatedBody=${!!translatedBody}, messages=${!!translatedBody?.messages}, isArray=${Array.isArray(translatedBody?.messages)}`
);
}

// Resolve executor with optional upstream proxy (CLIProxyAPI) routing.
Expand Down Expand Up @@ -1840,7 +1856,8 @@ export async function handleChatCore({
const newCredentials = (await refreshWithRetry(
() => executor.refreshCredentials(credentials, log),
3,
log
log,
provider // Explicitly pass the provider to avoid universally tripping the "unknown" circuit breaker
)) as null | {
accessToken?: string;
copilotToken?: string;
Expand Down
2 changes: 1 addition & 1 deletion open-sse/handlers/responseSanitizer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -464,7 +464,7 @@ function sanitizeResponsesOutputItem(item: unknown, index: number): JsonRecord |
text: collapseExcessiveNewlines(toString(partRecord.text) || ""),
};
})
.filter((part): part is JsonRecord => part !== null)
.filter((part): part is { type: string; text: string } => part !== null)
: [];

return {
Expand Down
11 changes: 10 additions & 1 deletion open-sse/services/bailianQuotaFetcher.ts
Original file line number Diff line number Diff line change
Expand Up @@ -94,7 +94,16 @@ function getAuthKey(
}

function getHost(): string {
return process.env.ALIBABA_CODING_PLAN_HOST || BAILIAN_QUOTA_HOSTS.international;
const configuredHost = process.env.ALIBABA_CODING_PLAN_HOST?.trim();
if (!configuredHost) {
return BAILIAN_QUOTA_HOSTS.international;
}

if (/^https?:\/\//i.test(configuredHost)) {
return configuredHost;
}

return `https://${configuredHost}`;
}

function getQuotaUrl(): string {
Expand Down
22 changes: 15 additions & 7 deletions open-sse/services/claudeCodeCompatible.ts
Original file line number Diff line number Diff line change
Expand Up @@ -437,11 +437,16 @@ function buildClaudeCodeCompatibleMessages(messages: MessageLike[]) {
.filter(
(
message
): message is { role: "user" | "assistant"; content: Array<Record<string, unknown>> } =>
!!message && message.content.length > 0
): message is {
role: "user" | "assistant";
content: Array<{ type: string; text: string }>;
} => !!message && message.content.length > 0
);

const merged: Array<{ role: "user" | "assistant"; content: Array<Record<string, unknown>> }> = [];
const merged: Array<{
role: "user" | "assistant";
content: Array<{ type: string; text: string }>;
}> = [];

for (const message of converted) {
const last = merged[merged.length - 1];
Expand Down Expand Up @@ -575,7 +580,7 @@ function buildClaudeCodeCompatibleSystemBlocks({
for (const systemBlock of customSystemBlocks) {
const preparedBlock = { ...systemBlock };
if (!preserveCacheControl) {
delete preparedBlock.cache_control;
delete preparedBlock["cache_control"];
}
blocks.push(preparedBlock);
}
Expand Down Expand Up @@ -700,9 +705,12 @@ function prepareClaudeCodeCompatibleBody(
const prepared = prepareClaudeRequest(
{
system: normalizeClaudeSystemInput(claudeBody.system),
messages: normalizeClaudeMessageInput(claudeBody.messages),
messages: normalizeClaudeMessageInput(claudeBody.messages) as Array<{
role?: string;
content?: string | Array<Record<string, unknown>>;
}>,
tools: normalizeClaudeToolInput(claudeBody.tools),
thinking: readRecord(claudeBody.thinking) || claudeBody.thinking,
thinking: (readRecord(claudeBody.thinking) || null) as Record<string, unknown> | null,
},
CLAUDE_CODE_COMPATIBLE_PREFIX,
true
Expand Down Expand Up @@ -735,7 +743,7 @@ function normalizeClaudeMessageInput(messages: unknown) {
content: normalizeClaudeContentInput(record.content),
};
})
.filter((message): message is Record<string, unknown> => !!message);
.filter((message): message is Record<string, unknown> & { content: unknown } => !!message);
}

function normalizeClaudeToolInput(tools: unknown) {
Expand Down
41 changes: 34 additions & 7 deletions open-sse/services/contextManager.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
*/

import { REGISTRY } from "../config/providerRegistry.ts";
import { getModelContextLimit } from "../../src/lib/modelCapabilities";
import { getModelContextLimit } from "../../src/lib/modelCapabilities.ts";

// Default token limits per provider (fallbacks when not in registry)
const DEFAULT_LIMITS: Record<string, number> = {
Expand Down Expand Up @@ -34,6 +34,16 @@ function getEnvOverride(provider: string): number | null {
return null;
}

// Reserve tokens override from environment variable
function getReserveTokensOverride(): number | null {
const envValue = process.env.CONTEXT_RESERVE_TOKENS;
if (envValue) {
const parsed = parseInt(envValue, 10);
if (!isNaN(parsed) && parsed > 0) return parsed;
}
return null;
}

// Rough chars-per-token ratio for quick estimation
const CHARS_PER_TOKEN = 4;

Expand Down Expand Up @@ -109,8 +119,12 @@ export function compressContext(
const provider = options.provider || "default";
const maxTokens =
options.maxTokens || getTokenLimit(provider, (body.model as string) || options.model || null);
const reserveTokens = options.reserveTokens || 16000; // Reserve for response
const targetTokens = maxTokens - reserveTokens;
const defaultReserveTokens = Math.min(16000, Math.max(256, Math.floor(maxTokens * 0.15)));
const reserveTokens = Math.min(
options.reserveTokens ?? getReserveTokensOverride() ?? defaultReserveTokens,
Math.max(0, maxTokens - 1)
);
const targetTokens = Math.max(0, maxTokens - reserveTokens);

let messages = [...body.messages];
let currentTokens = estimateTokens(JSON.stringify(messages));
Expand Down Expand Up @@ -216,10 +230,23 @@ function compressThinking(messages: Record<string, unknown>[]) {

// Remove thinking XML tags from string content
if (typeof msg.content === "string") {
const cleaned = msg.content
.replace(/<thinking>.*?<\/thinking>/gs, "")
.replace(/<antThinking>.*?<\/antThinking>/gs, "")
.trim();
let cleaned = msg.content;
for (const [start, end] of [
["<thinking>", "</thinking>"],
["<antThinking>", "</antThinking>"],
]) {
while (true) {
const s = cleaned.indexOf(start);
if (s === -1) break;
const e = cleaned.indexOf(end, s + start.length);
if (e === -1) {
cleaned = cleaned.slice(0, s);
break;
}
cleaned = cleaned.slice(0, s) + cleaned.slice(e + end.length);
}
}
cleaned = cleaned.trim();
return { ...msg, content: cleaned || "[thinking compressed]" };
}

Expand Down
10 changes: 8 additions & 2 deletions open-sse/services/provider.ts
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,10 @@ export function isClaudeCodeCompatible(provider) {
return typeof provider === "string" && provider.startsWith(CLAUDE_CODE_COMPATIBLE_PREFIX);
}

export function getOpenAICompatibleType(provider, providerSpecificData = null) {
export function getOpenAICompatibleType(
provider,
providerSpecificData: Record<string, unknown> | null = null
) {
if (!isOpenAICompatible(provider)) return "chat";
const configuredType =
providerSpecificData &&
Expand Down Expand Up @@ -277,7 +280,10 @@ export function buildProviderUrl(
}
// Custom URL builder (e.g. gemini, gemini-cli)
if (entry.urlBuilder) {
return entry.urlBuilder(entry.baseUrl, model, stream);
const baseUrl = entry.baseUrl || config.baseUrl;
if (baseUrl) {
return entry.urlBuilder(baseUrl, model, stream);
}
}
// URL suffix (e.g. claude: ?beta=true)
if (entry.urlSuffix) {
Expand Down
Loading