Skip to content
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -557,7 +557,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute
- **🧠 Memory you control** — off by default, opt-in int8 vector quantization + typed decay, per-request `x-omniroute-no-memory`. → [Memory](docs/frameworks/MEMORY.md)
- **🛡️ Security** — prompt-injection guard on every LLM route (red-team suite), opt-in credential-masking guardrail (redacts leaked API keys/secrets in both directions), free DuckDuckGo last-resort web search, and an optional OIDC login gate for the dashboard (password login always stays available). → [Guardrails](docs/security/GUARDRAILS.md)
- **🖼️ New endpoints** — `/v1/ocr` (Mistral OCR) and `/v1/audio/translations` (Whisper-style) round out the media surface. → [API Reference](docs/reference/API_REFERENCE.md)
- **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Google Imagen, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md)
- **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md)
- **🌍 Deployment & ops** — reverse-proxy `basePath`, browser-language auto-detect, per-key device tracking, root-less MITM trust, zh-TW localization. → [Environment](docs/reference/ENVIRONMENT.md)
- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **340-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md)
- **📡 Routing transparency** — every response carries an `X-OmniRoute-Decision` header naming the strategy/provider/latency that served it, a new `cache-optimized` combo strategy + Auto-Combo `cacheAffinity` factor route repeat requests back to the connection holding the cached prefix, and a read-only `/v1/auto-combo/{channel}/candidates` endpoint exposes an `auto/*` channel's live candidate pool. → [Auto-Combo](docs/routing/AUTO-COMBO.md)
Expand Down
115 changes: 34 additions & 81 deletions open-sse/config/agyModels.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,8 +2,8 @@
//
// These models are pinned from the live `:fetchAvailableModels` endpoint
// (https://daily-cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels) using a
// real `agy` consumer-OAuth token. The public catalog exposes the upstream Gemini 3.6
// and 3.5 Flash ids verbatim; the shared Antigravity executor dispatches them unchanged.
// real `agy` consumer-OAuth token. The public catalog exposes the upstream Gemini 3.7
// Flash ids verbatim; the shared Antigravity executor dispatches them unchanged.
//
// The `agy` provider reuses the `antigravity` executor/translator (identical backend),
// but keeps its own catalog so the CLI and IDE model surfaces can evolve independently.
Expand All @@ -12,48 +12,29 @@
// they are not chat-callable.

export const AGY_PUBLIC_MODELS = Object.freeze([
// Gemini 3.6 Flash tiers. The live endpoint selects High by default and advertises
// all three ids to both the IDE 2.1.1 and CLI 1.1.x clients.
// Gemini 3.7 Flash tiers. The live endpoint selects High by default and advertises
// all three ids to both the IDE 2.5.5 and CLI 1.1.x clients.
{
id: "gemini-3.6-flash-high",
name: "Gemini 3.6 Flash (High)",
id: "gemini-3.7-flash-high",
name: "Gemini 3.7 Flash (High)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.6-flash-medium",
name: "Gemini 3.6 Flash (Medium)",
id: "gemini-3.7-flash-medium",
name: "Gemini 3.7 Flash (Medium)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.6-flash-low",
name: "Gemini 3.6 Flash (Low)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
// Claude (Antigravity backend).
{
id: "claude-opus-4-6-thinking",
name: "Claude Opus 4.6 (Thinking)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "claude-sonnet-4-6",
name: "Claude Sonnet 4.6 (Thinking)",
id: "gemini-3.7-flash-low",
name: "Gemini 3.7 Flash (Low)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
Expand All @@ -80,74 +61,31 @@ export const AGY_PUBLIC_MODELS = Object.freeze([
toolCalling: true,
},
{
id: "gemini-3-flash-agent",
name: "Gemini 3.5 Flash (High)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.5-flash-low",
name: "Gemini 3.5 Flash (Medium)",
id: "gemini-3.1-flash-lite",
name: "Gemini 3.1 Flash Lite",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
maxOutputTokens: 65535,
toolCalling: true,
},
// Claude (Antigravity backend).
{
id: "gemini-3.5-flash-extra-low",
name: "Gemini 3.5 Flash (Low)",
id: "claude-opus-4-6-thinking",
name: "Claude Opus 4.6 (Thinking)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
// Gemini 3.7 Flash: single callable public model (upstream exposes only
// gemini-3.7-flash-tiered; suffixed tier ids 404). One entry so it does not
// collide under the #3696 public-id uniqueness invariant.
{
id: "gemini-3.7-flash",
name: "Gemini 3.7 Flash",
id: "claude-sonnet-4-6",
name: "Claude Sonnet 4.6 (Thinking)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.1-flash-lite",
name: "Gemini 3.1 Flash Lite",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
// Gemini 2.5
{
id: "gemini-2.5-flash-thinking",
name: "Gemini 2.5 Flash Thinking",
contextLength: 1048576,
maxOutputTokens: 65535,
supportsReasoning: true,
toolCalling: true,
},
{
id: "gemini-2.5-flash",
name: "Gemini 2.5 Flash",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
{
id: "gemini-2.5-flash-lite",
name: "Gemini 2.5 Flash Lite",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
// GPT-OSS
{
id: "gpt-oss-120b-medium",
Expand All @@ -161,6 +99,21 @@ export const AGY_PUBLIC_MODELS = Object.freeze([

const AGY_PUBLIC_MODEL_IDS = new Set(AGY_PUBLIC_MODELS.map((model) => model.id));
const AGY_NON_CHAT_MODEL_IDS = new Set(["tab_flash_lite_preview", "tab_jump_flash_lite_preview"]);
const AGY_RETIRED_MODEL_IDS = new Set([
"gemini-3.6-flash-high",
"gemini-3.6-flash-medium",
"gemini-3.6-flash-low",
"gemini-3-flash-agent",
"gemini-3.5-flash-extra-low",
"gemini-3.5-flash-low",
"gemini-3.5-flash-high",
"gemini-3.5-flash-medium",
"gemini-3.5-flash-preview",
"gemini-2.5-pro",
"gemini-2.5-flash-thinking",
"gemini-2.5-flash",
"gemini-2.5-flash-lite",
]);

const AGY_CLIENT_VISIBLE_MODEL_NAMES = Object.freeze(
AGY_PUBLIC_MODELS.reduce<Record<string, string>>((acc, model) => {
Expand All @@ -178,5 +131,5 @@ export function isUserCallableAgyModelId(modelId: string): boolean {
}

export function isDiscoverableAgyModelId(modelId: string): boolean {
return !!modelId && !AGY_NON_CHAT_MODEL_IDS.has(modelId);
return !!modelId && !AGY_NON_CHAT_MODEL_IDS.has(modelId) && !AGY_RETIRED_MODEL_IDS.has(modelId);
}
140 changes: 32 additions & 108 deletions open-sse/config/antigravityModelAliases.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([
// Gemini 3.7 Flash tiers listed by the current official Antigravity model catalog
// alongside the existing Gemini 3.6 tiers. Keep the upstream model ids unchanged so
// discovery and execution address the same models selected by the native client.
// Gemini 3.7 Flash tiers listed by the current official Antigravity model catalog.
// Keep the upstream model ids unchanged so discovery and execution address the same
// models selected by the native client.
{
id: "gemini-3.7-flash-high",
name: "Gemini 3.7 Flash (High)",
Expand All @@ -20,51 +20,9 @@ export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([
supportsVision: true,
toolCalling: true,
},
// Gemini 3.6 Flash tiers retained alongside the newer Gemini 3.7 tiers.
{
id: "gemini-3.6-flash-high",
name: "Gemini 3.6 Flash (High)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.6-flash-medium",
name: "Gemini 3.6 Flash (Medium)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.6-flash-low",
name: "Gemini 3.6 Flash (Low)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
// Claude (Antigravity backend). The `agy` provider already ships these from the live
// :fetchAvailableModels probe (see agyModels.ts) and discussion #3184 confirmed they
// are user-callable through the `antigravity` OAuth provider too — same backend.
// `antigravity/claude-opus-4-6-thinking` and `antigravity/claude-sonnet-4-6` both work.
// They are upstream IDs, so no alias remapping is required.
{
id: "claude-opus-4-6-thinking",
name: "Claude Opus 4.6 (Thinking)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "claude-sonnet-4-6",
name: "Claude Sonnet 4.6 (Thinking)",
id: "gemini-3.7-flash-low",
name: "Gemini 3.7 Flash (Low)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
Expand Down Expand Up @@ -92,78 +50,36 @@ export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([
supportsVision: true,
toolCalling: true,
},
// Gemini 3.5 Flash tiers exposed by Antigravity's model selector. Public ids match
// fetchAvailableModels and are forwarded upstream unchanged:
// High -> gemini-3-flash-agent (displayName: Gemini 3.5 Flash (High))
// Medium -> gemini-3.5-flash-low (displayName: Gemini 3.5 Flash (Medium))
// Low -> gemini-3.5-flash-extra-low (displayName: Gemini 3.5 Flash (Low))
{
id: "gemini-3-flash-agent",
name: "Gemini 3.5 Flash (High)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.5-flash-low",
name: "Gemini 3.5 Flash (Medium)",
id: "gemini-3.1-flash-lite",
name: "Gemini 3.1 Flash Lite",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
maxOutputTokens: 65535,
toolCalling: true,
},
// Claude (Antigravity backend). The `agy` provider already ships these from the live
// :fetchAvailableModels probe (see agyModels.ts) and discussion #3184 confirmed they
// are user-callable through the `antigravity` OAuth provider too — same backend.
// `antigravity/claude-opus-4-6-thinking` and `antigravity/claude-sonnet-4-6` both work.
// They are upstream IDs, so no alias remapping is required.
{
id: "gemini-3.5-flash-extra-low",
name: "Gemini 3.5 Flash (Low)",
id: "claude-opus-4-6-thinking",
name: "Claude Opus 4.6 (Thinking)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
// Gemini 3.7 Flash: Antigravity's live catalog exposes a single upstream id
// gemini-3.7-flash-tiered; the suffixed tier ids 404 upstream. Kept as one
// callable public model so it does not collide with the #3696 uniqueness invariant.
{
id: "gemini-3.7-flash",
name: "Gemini 3.7 Flash",
id: "claude-sonnet-4-6",
name: "Claude Sonnet 4.6 (Thinking)",
contextLength: 1048576,
maxOutputTokens: 65536,
supportsReasoning: true,
supportsVision: true,
toolCalling: true,
},
{
id: "gemini-3.1-flash-lite",
name: "Gemini 3.1 Flash Lite",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
{
id: "gemini-2.5-flash-thinking",
name: "Gemini 2.5 Flash Thinking",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
{
id: "gemini-2.5-flash",
name: "Gemini 2.5 Flash",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
{
id: "gemini-2.5-flash-lite",
name: "Gemini 2.5 Flash Lite",
contextLength: 1048576,
maxOutputTokens: 65535,
toolCalling: true,
},
{
id: "gpt-oss-120b-medium",
name: "GPT-OSS 120B (Medium)",
Expand All @@ -175,12 +91,6 @@ export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([
]);

export const ANTIGRAVITY_MODEL_ALIASES = Object.freeze({
// Gemini 3.7 Flash: the live catalog (fetchAvailableModels on daily-cloudcode-pa)
// exposes a single upstream id `gemini-3.7-flash-tiered`; the agy CLI maps all
// display tiers (high/medium/low) to it. Verified 200 OK with thinking_level and
// thinkingBudget configs. The suffixed ids 404 upstream ("Requested entity was not found").
// Exposed as ONE callable model (see #3696: public ids must be unique upstream ids).
"gemini-3.7-flash": "gemini-3.7-flash-tiered",
// gemini-3.1-pro-low is not aliased: the upstream accepts it verbatim.
// gemini-3.1-pro-high: the discovery slot returns HTTP 400 on v1internal;
// the live upstream id is gemini-pro-agent (see ANTIGRAVITY_PUBLIC_MODELS).
Expand Down Expand Up @@ -249,10 +159,19 @@ const ANTIGRAVITY_NON_CHAT_MODEL_IDS = new Set([
const ANTIGRAVITY_RETIRED_MODEL_IDS = new Set([
"gemini-3-pro-preview",
"gemini-3.1-pro",
"gemini-3.6-flash-high",
"gemini-3.6-flash-medium",
"gemini-3.6-flash-low",
"gemini-3-flash-agent",
"gemini-3.5-flash-extra-low",
"gemini-3.5-flash-low",
"gemini-3.5-flash-high",
"gemini-3.5-flash-medium",
"gemini-3.5-flash-preview",
"gemini-2.5-pro",
"gemini-2.5-flash-thinking",
"gemini-2.5-flash",
"gemini-2.5-flash-lite",
"gemini-2.5-computer-use-preview-10-2025",
]);

Expand Down Expand Up @@ -281,7 +200,12 @@ const ANTIGRAVITY_DROPPED_QUOTA_BUCKETS = new Set<string>([
*/
export function toClientAntigravityQuotaModelId(modelId: string): string | null {
if (!modelId) return null;
if (ANTIGRAVITY_DROPPED_QUOTA_BUCKETS.has(modelId)) return null;
if (
ANTIGRAVITY_DROPPED_QUOTA_BUCKETS.has(modelId) ||
ANTIGRAVITY_RETIRED_MODEL_IDS.has(modelId)
) {
return null;
}
return toClientAntigravityModelId(modelId);
}

Expand Down
Loading
Loading