From 38ba856b6986455d93082738271e2c5f986dd40a Mon Sep 17 00:00:00 2001 From: pacocartones Date: Sun, 23 Aug 2026 01:09:08 +0200 Subject: [PATCH 01/29] fix(resilience): route chat by per-connection synced model inventory (#11089) Multi-host self-hosted providers (ollama-local, lm-studio, vllm, ...) keep a per-connection synced inventory, but getProviderCredentials only ever consulted the manual excludedModels denylist. A request for a model that only one host advertises could therefore be routed to a host that never had it, producing a spurious model-not-found instead of pinning to the host that does. Adds isModelAdvertisedByConnection to connectionModelRules (literal id match, same candidate semantics as the denylist, fails open on an empty inventory) and an additional predicate in the availableConnections filter, scoped to SELF_HOSTED_CHAT_PROVIDER_IDS. --- .../11089-chat-routing-synced-inventory.md | 1 + src/domain/connectionModelRules.ts | 23 +++ src/sse/services/auth.ts | 80 +++++++- ...hat-routing-synced-inventory-11089.test.ts | 184 ++++++++++++++++++ 4 files changed, 287 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/11089-chat-routing-synced-inventory.md create mode 100644 tests/unit/chat-routing-synced-inventory-11089.test.ts diff --git a/changelog.d/fixes/11089-chat-routing-synced-inventory.md b/changelog.d/fixes/11089-chat-routing-synced-inventory.md new file mode 100644 index 00000000000..922b96a659b --- /dev/null +++ b/changelog.d/fixes/11089-chat-routing-synced-inventory.md @@ -0,0 +1 @@ +- **fix(resilience):** filter chat connection selection by each connection's *synced* model inventory on multi-host self-hosted providers (`ollama-local`, `lm-studio`, `vllm`, …), so a request for a model only one host advertises is pinned to that host instead of failing over onto a host that never had it ([#11089](https://github.com/diegosouzapw/OmniRoute/issues/11089)) diff --git a/src/domain/connectionModelRules.ts b/src/domain/connectionModelRules.ts index 316ade72d80..7831bbc84b6 100644 --- a/src/domain/connectionModelRules.ts +++ b/src/domain/connectionModelRules.ts @@ -80,3 +80,26 @@ export function hasEligibleConnectionForModel( (connection) => !isModelExcludedByConnection(modelId, connection?.providerSpecificData) ); } + +/** + * #11089: does this connection's *synced* inventory advertise the model? + * + * Unlike `excludedModels` (a manually maintained denylist) this reads the + * per-connection catalog written by model discovery, so a multi-host local + * provider never routes a model to a host that never had it. Ids are matched + * with the same candidate semantics as the denylist (provider prefix and the + * `[1m]` extended-context suffix are tolerated), but never as wildcard + * patterns — a synced id is a literal. + * + * Fails OPEN on an empty inventory: a host that has not been synced yet is + * "unknown", not "does not have it". + */ +export function isModelAdvertisedByConnection( + modelId: unknown, + advertisedModelIds: ReadonlySet | null | undefined +): boolean { + if (!advertisedModelIds || advertisedModelIds.size === 0) return true; + if (typeof modelId !== "string" || modelId.trim().length === 0) return true; + + return getModelMatchCandidates(modelId).some((candidate) => advertisedModelIds.has(candidate)); +} diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index b1df29552a4..dca29cad216 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -106,8 +106,17 @@ import { resolveProviderId, NOAUTH_PROVIDERS, WEB_COOKIE_PROVIDERS, + isSelfHostedChatProvider, } from "@/shared/constants/providers"; -import { isModelExcludedByConnection } from "@/domain/connectionModelRules"; +import { + isModelExcludedByConnection, + isModelAdvertisedByConnection, +} from "@/domain/connectionModelRules"; +import { + getSyncedAvailableModelsByConnection, + SYNCED_AVAILABLE_MODELS_MALFORMED, + type SyncedAvailableModelsByConnection, +} from "@/lib/db/models"; import { isFreeModel } from "@/shared/utils/freeModels"; import { applySessionAffinityPin, @@ -1160,6 +1169,54 @@ function materializeConnection( }; } +/** + * #11089: load the per-connection synced model inventory for self-hosted chat + * providers so connection selection can drop hosts that never advertised the + * requested model. + * + * Scoped to SELF_HOSTED_CHAT_PROVIDER_IDS: those are the providers where one + * provider id fans out to several independent hosts with genuinely different + * inventories. Hosted providers share one catalog per provider, so filtering + * there would only add a DB read. + * + * Returns an empty map (= no filtering) when there is no model to match, when + * no candidate is self-hosted, or when the persisted rows are malformed — a + * partial read must never silently shrink the pool. + */ +async function loadAdvertisedModelsForSelfHostedConnections( + connections: ProviderConnectionView[], + requestedModel: string | null +): Promise>> { + const advertised = new Map>(); + if (!requestedModel) return advertised; + + const selfHostedProviders = new Set( + connections + .map((c) => c.provider) + .filter((p): p is string => typeof p === "string" && isSelfHostedChatProvider(p)) + ); + if (selfHostedProviders.size === 0) return advertised; + + await Promise.all( + [...selfHostedProviders].map(async (providerId) => { + let byConnection: SyncedAvailableModelsByConnection; + try { + byConnection = await getSyncedAvailableModelsByConnection(providerId); + } catch { + return; + } + // Malformed persisted rows: fail open for the whole provider. + if (byConnection[SYNCED_AVAILABLE_MODELS_MALFORMED]) return; + for (const [connectionId, models] of Object.entries(byConnection)) { + if (!Array.isArray(models) || models.length === 0) continue; + advertised.set(connectionId, new Set(models.map((m) => m.id))); + } + }) + ); + + return advertised; +} + /** * Get provider credentials from localDb * Filters out unavailable accounts and returns the selected account based on strategy @@ -1435,6 +1492,14 @@ export async function getProviderCredentials( let modelLockedCount = 0; let familyLockedCount = 0; const connectionFilterStatus = new Map(); + // #11089: multi-host self-hosted providers keep a per-connection synced + // inventory. Without it, a request can be routed to a host that never had + // the model, producing a spurious model-not-found instead of pinning to + // the host that does. Empty map = no inventory known = no filtering. + const advertisedModelsByConnection = await loadAdvertisedModelsForSelfHostedConnections( + connections, + requestedModel + ); // Filter out unavailable accounts and excluded connection let availableConnections = connections.filter((c) => { if (excludedConnectionIds.has(c.id)) { @@ -1445,6 +1510,13 @@ export async function getProviderCredentials( connectionFilterStatus.set(c.id, "modelExcluded"); return false; } + if ( + requestedModel && + !isModelAdvertisedByConnection(requestedModel, advertisedModelsByConnection.get(c.id)) + ) { + connectionFilterStatus.set(c.id, "modelNotAdvertised"); + return false; + } if (!allowSuppressedConnections) { if (!allowRateLimitedConnections && isAccountUnavailable(c.rateLimitedUntil)) { connectionFilterStatus.set(c.id, "rateLimited"); @@ -1510,6 +1582,7 @@ export async function getProviderCredentials( const codexScopeLimited = status === "codexScopeLimited"; const modelLocked = status === "modelLocked"; const modelExcluded = status === "modelExcluded"; + const modelNotAdvertised = status === "modelNotAdvertised"; if (excluded || rateLimited) { log.debug( "AUTH", @@ -1520,6 +1593,11 @@ export async function getProviderCredentials( "AUTH", ` → ${c.id?.slice(0, 8)} | excluded by per-account model rule for ${requestedModel}` ); + } else if (modelNotAdvertised) { + log.debug( + "AUTH", + ` → ${c.id?.slice(0, 8)} | synced inventory does not advertise ${requestedModel}` + ); } else if (terminalStatus) { log.debug( "AUTH", diff --git a/tests/unit/chat-routing-synced-inventory-11089.test.ts b/tests/unit/chat-routing-synced-inventory-11089.test.ts new file mode 100644 index 00000000000..441b14893b6 --- /dev/null +++ b/tests/unit/chat-routing-synced-inventory-11089.test.ts @@ -0,0 +1,184 @@ +/** + * tests/unit/chat-routing-synced-inventory-11089.test.ts + * + * #11089 — Chat routing ignores per-connection model inventory on multi-host + * local providers. + * + * One self-hosted provider (`ollama-local`) with TWO connections pointing at + * different hosts and DISJOINT synced inventories: + * + * studio (priority 1) → gemma3:4b, flux2-klein:9b + * jetson (priority 2) → gemma3:4b + * + * `getProviderCredentials` only ever consulted the manual `excludedModels` + * denylist, never the synced inventory persisted per connection, so a request + * for `flux2-klein:9b` could land on jetson — a host that never had the model. + * + * Cases: + * 1. Model advertised by only one connection → the other is never selected. + * 2. Higher-priority host cooling → must NOT preemptively fall to a host that + * lacks the model. + * 3. The advertising connection stays selectable. + * 4. Model advertised by both → both remain eligible (no over-filtering). + * 5. Provider with NO synced inventory at all → fail open, selection unchanged. + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-chat-synced-11089-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const modelsDb = await import("../../src/lib/db/models.ts"); +const auth = await import("../../src/sse/services/auth.ts"); + +const PROVIDER = "ollama-local"; + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +async function createConnection(data: Record): Promise { + const created = (await providersDb.createProviderConnection(data)) as { id: string }; + return created.id; +} + +/** The connection id the selector handed back, or null if it returned no account. */ +function selectedConnectionId(selected: unknown): string | null { + if (!selected || typeof selected !== "object") return null; + const id = (selected as { connectionId?: unknown }).connectionId; + return typeof id === "string" ? id : null; +} + +/** Create the two-host ollama-local topology from the issue report. */ +async function seedTwoHosts(options: { studioRateLimitedUntil?: string } = {}) { + const studioId = await createConnection({ + provider: PROVIDER, + authType: "none", + name: "Mac Studio", + baseUrl: "http://studio.lan:11434/v1", + priority: 1, + isActive: true, + }); + const jetsonId = await createConnection({ + provider: PROVIDER, + authType: "none", + name: "Jetson", + baseUrl: "http://jetson.lan:11434/v1", + priority: 2, + isActive: true, + }); + + await modelsDb.replaceSyncedAvailableModelsForConnection(PROVIDER, studioId, [ + { id: "gemma3:4b", name: "gemma3:4b" }, + { id: "flux2-klein:9b", name: "flux2-klein:9b" }, + ]); + await modelsDb.replaceSyncedAvailableModelsForConnection(PROVIDER, jetsonId, [ + { id: "gemma3:4b", name: "gemma3:4b" }, + ]); + + if (options.studioRateLimitedUntil) { + await providersDb.updateProviderConnection(studioId, { + rateLimitedUntil: options.studioRateLimitedUntil, + }); + } + + return { studioId, jetsonId }; +} + +test("#11089 selects only the host whose synced inventory advertises the model", async () => { + await resetStorage(); + const { studioId, jetsonId } = await seedTwoHosts(); + + // Exclude studio to force the selector to look elsewhere. Jetson does not + // advertise flux2-klein:9b, so it must NOT be handed back. + const selected = await auth.getProviderCredentials(PROVIDER, studioId, null, "flux2-klein:9b"); + + assert.notEqual( + selectedConnectionId(selected), + jetsonId, + "jetson never synced flux2-klein:9b and must not be selected for it" + ); +}); + +test("#11089 does not preemptively fail over to a host lacking the model when the owner is cooling", async () => { + await resetStorage(); + const coolingUntil = new Date(Date.now() + 10 * 60 * 1000).toISOString(); + const { jetsonId } = await seedTwoHosts({ studioRateLimitedUntil: coolingUntil }); + + const selected = await auth.getProviderCredentials(PROVIDER, null, null, "flux2-klein:9b"); + + assert.notEqual( + selectedConnectionId(selected), + jetsonId, + "a cooling studio must surface a cooldown, not silently route to a host without the model" + ); +}); + +test("#11089 keeps the connection that does advertise the model selectable", async () => { + await resetStorage(); + const { studioId } = await seedTwoHosts(); + + const selected = await auth.getProviderCredentials(PROVIDER, null, null, "flux2-klein:9b"); + + assert.equal( + selectedConnectionId(selected), + studioId, + "studio advertises flux2-klein:9b and must be selected" + ); +}); + +test("#11089 a model advertised by every host leaves both connections eligible", async () => { + await resetStorage(); + const { studioId, jetsonId } = await seedTwoHosts(); + + const first = await auth.getProviderCredentials(PROVIDER, null, null, "gemma3:4b"); + assert.equal( + selectedConnectionId(first), + studioId, + "fill-first prefers priority 1 for a shared model" + ); + + // Excluding studio (the normal account-fallback path) must still reach jetson, + // because jetson genuinely advertises gemma3:4b. + const second = await auth.getProviderCredentials(PROVIDER, studioId, null, "gemma3:4b"); + assert.equal( + selectedConnectionId(second), + jetsonId, + "jetson advertises gemma3:4b and must remain a valid failover" + ); +}); + +test("#11089 fails open when the provider has no synced inventory at all", async () => { + await resetStorage(); + + const connectionId = await createConnection({ + provider: PROVIDER, + authType: "none", + baseUrl: "http://127.0.0.1:11434/v1", + priority: 1, + isActive: true, + }); + + // No replaceSyncedAvailableModelsForConnection call: discovery never ran. + // Routing must behave exactly as before rather than filtering everything out. + const selected = await auth.getProviderCredentials(PROVIDER, null, null, "never-synced-model"); + + assert.equal( + selectedConnectionId(selected), + connectionId, + "an unsynced provider must not be filtered to zero candidates" + ); +}); From f3875759aced242e0f8dbd2f6ec0c9d904e56077 Mon Sep 17 00:00:00 2001 From: ggdayup Date: Sun, 23 Aug 2026 08:51:16 +0800 Subject: [PATCH 02/29] fix(memory): treat TokenRouter as system-must-be-first (live HTTP 400 confirmed) (#11114) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board over tip 92ef3c71: static gates clean (changelog, file-size, complexity 2624<=2774, cognitive 1182<=1223, dead-code 411<=416), typecheck:core clean, focused tests green. Companion to #11113 (consumer side): tokenrouter joins BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST — memory-system-first-6135 suite green. Live-confirmed 400 class documented in the body. Thank you @ggdayup! --- src/lib/memory/injection.ts | 8 ++++++-- tests/unit/memory-system-first-6135.test.ts | 5 +++++ 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/src/lib/memory/injection.ts b/src/lib/memory/injection.ts index 9fdd2bbbf10..d4d8ead7f7c 100644 --- a/src/lib/memory/injection.ts +++ b/src/lib/memory/injection.ts @@ -65,7 +65,11 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * * Populated with the Xiaomi MiMo endpoint (provider id `xiaomi-mimo`, registry * alias `mimo`, serving mimo-v2.5) confirmed live to 400 on a non-first system - * message. Add other providers here only when they are documented as strict. + * message, and the TokenRouter gateway (provider id `tokenrouter`), confirmed + * live on 2026-08-22 to reject mid-array system messages — including the + * compression notice spliced by purifyHistory() before that splice was fixed to + * merge into the leading system message. Add other providers here only when + * they are documented as strict. * * Self-hosted deployments can extend this list without a source change via * OMNIROUTE_STRICT_SYSTEM_PROVIDERS (comma-separated provider ids, @@ -73,7 +77,7 @@ export function providerSupportsSystemMessage(provider: string | null | undefine * self-hosted Qwen3.5+/3.6 model, whose chat template enforces the same * single-leading-system-message constraint as xiaomi-mimo. */ -const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo"]); +const BUILTIN_PROVIDERS_SYSTEM_MUST_BE_FIRST = new Set(["xiaomi-mimo", "mimo", "tokenrouter"]); /** * Parses OMNIROUTE_STRICT_SYSTEM_PROVIDERS into a normalized id list. diff --git a/tests/unit/memory-system-first-6135.test.ts b/tests/unit/memory-system-first-6135.test.ts index 7104007ae64..339f83792cf 100644 --- a/tests/unit/memory-system-first-6135.test.ts +++ b/tests/unit/memory-system-first-6135.test.ts @@ -49,6 +49,11 @@ describe("injectMemory system-must-be-first (#6135)", () => { it("flags xiaomi-mimo (and alias mimo) as system-must-be-first", () => { assert.equal(systemMessageMustBeFirst("xiaomi-mimo"), true); assert.equal(systemMessageMustBeFirst("mimo"), true); + // tokenrouter: confirmed live 2026-08-22 — mid-array system message + // (e.g. the purifyHistory compression notice) -> HTTP 400 + // "System message must be at the beginning". + assert.equal(systemMessageMustBeFirst("tokenrouter"), true); + assert.equal(systemMessageMustBeFirst("TokenRouter"), true); // case-insensitive // default: unlisted providers keep current (non-first-constrained) behavior assert.equal(systemMessageMustBeFirst("anthropic"), false); assert.equal(systemMessageMustBeFirst(null), false); From b3844550d009e34c891bc629ca4193c4a09f1c5d Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Sun, 23 Aug 2026 02:51:19 +0200 Subject: [PATCH 03/29] fix(api): refuse creating a routing combo without any model (#11162) Validated on the combined batch board over tip 92ef3c71: static gates clean (changelog, file-size, complexity 2624<=2774, cognitive 1182<=1223, dead-code 411<=416), typecheck:core clean, focused tests green. Combo without models is now refused at the schema boundary (API 400), the CLI flags it, and openapi.yaml matches the real contract (phantom props removed). combo-* suites + cli-combo-create-models green on the board. Closes #10954. Thank you @maxmad64bis! --- bin/cli/commands/combo.mjs | 6 ++++ .../11162-combo-create-requires-model.md | 1 + config/quality/quality-baseline.json | 5 +-- docs/openapi.yaml | 25 +++++++------- src/shared/validation/schemas/combo.ts | 7 ++-- .../cli-combo-create-models-10954.test.ts | 33 ++++++++++++++++++- tests/unit/combo-bracket-names.test.ts | 1 + tests/unit/combo-context-length.test.ts | 7 ++++ tests/unit/combo-empty-models.test.ts | 6 ++-- 9 files changed, 68 insertions(+), 23 deletions(-) create mode 100644 changelog.d/fixes/11162-combo-create-requires-model.md diff --git a/bin/cli/commands/combo.mjs b/bin/cli/commands/combo.mjs index 554c1c1d838..8d58cf73bd9 100644 --- a/bin/cli/commands/combo.mjs +++ b/bin/cli/commands/combo.mjs @@ -307,6 +307,12 @@ export async function runComboCreateCommand(name, strategy = "priority", opts = } const models = Array.isArray(opts.models) ? opts.models : []; + if (!models.length) { + console.error( + "combo create requires at least one target. Pass --models and/or repeat --model ." + ); + return 1; + } try { return await withRuntime(async ({ kind, api, db }) => { diff --git a/changelog.d/fixes/11162-combo-create-requires-model.md b/changelog.d/fixes/11162-combo-create-requires-model.md new file mode 100644 index 00000000000..228e6a9b20f --- /dev/null +++ b/changelog.d/fixes/11162-combo-create-requires-model.md @@ -0,0 +1 @@ +- **Combo create:** creating a routing combo without any model is now refused (`400`) — the CLI requires `--models`/`--model` on `combo create`, matching the dashboard which already rejected empty combos. diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index 95e35a5adec..604a5a8e19d 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -197,10 +197,11 @@ "_rebaseline_2026_08_09_v3850_release_close": "7666 -> 8045 (+379 gzip bytes, +4.9%). Release v3.8.50 close reconciliation measured twice with the real size-limit + @size-limit/file path on tip e0ce95c592. Per-entry measurements remain below their absolute budgets: omniroute.mjs 4380/15000, mcp-server.mjs 1195/5000, nodeRuntimeSupport.mjs 887/8000, reset-password.mjs 1583/6000. The growth accumulated through legitimate CLI/runtime work in this cycle, including global-install ESM alias resolution, Termux cache preparation, and MCP stdio startup hardening; no entrypoint is near its absolute ceiling. The direction:down ratchet stays blocking from this exact measured tip." }, "openapiBreaking": { - "value": 0, + "value": 4, "direction": "down", "dedicatedGate": true, - "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec." + "_note": "oasdiff breaking-change gate (Fase 9 Onda 0). Blocks any breaking change vs base spec.", + "_rebaseline_2026_08_22_combo_create_min1": "0 -> 4, split 3 own + 1 inherited. Docs-only alignment of components.schemas.ComboCreate with the request contract already enforced by the API since 638fc5fbd (combo create refuses an empty model list) and d5034ea52: `model`/`nodes` were phantom properties the server never accepted, and `models` (array, minItems 1) is the real required field. OWN findings (3, caused by this commit): removed `model`, removed `nodes`, added required `models` on POST /api/combos — spec-vs-server drift, not client-facing breakage, no working client could have relied on the removed shapes. INHERITED finding (1, NOT caused by this PR's code changes — pre-existing drift already present at parent d5034ea52): PATCH /api/combos/{id} request-body-added-required; that route's patch operation declares its own inline requestBody (required: true, bare object schema, docs/openapi.yaml ~2107-2118) and does not reference ComboCreate, so this finding exists independently of the ComboCreate alignment (same own-growth vs inherited-drift convention as _rebaseline_2026_07_20_aliasresolver_hook_split_7808). No code change in this PR; follow-up tracking = this change's PR description." }, "mutationScore.src/sse/services/auth.ts": { "value": 52.57, diff --git a/docs/openapi.yaml b/docs/openapi.yaml index e51b6091aea..73941f37ead 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -8893,12 +8893,20 @@ components: ComboCreate: type: object - required: [name, model] + required: [name, models] properties: name: type: string - model: - type: string + models: + type: array + minItems: 1 + items: + oneOf: + - type: string + description: "provider/model reference" + - type: object + description: "structured combo step (provider, model, weight, ...)" + additionalProperties: true strategy: type: string enum: @@ -8920,14 +8928,3 @@ components: - context-optimized - fusion default: priority - nodes: - type: array - items: - type: object - properties: - connectionId: - type: string - weight: - type: integer - priority: - type: integer diff --git a/src/shared/validation/schemas/combo.ts b/src/shared/validation/schemas/combo.ts index edeca47e729..db825e11cc9 100644 --- a/src/shared/validation/schemas/combo.ts +++ b/src/shared/validation/schemas/combo.ts @@ -321,7 +321,7 @@ export const createComboSchema = z .object({ name: comboNameSchema, description: z.string().max(2000).optional(), - models: z.array(comboModelEntry).optional().default([]), + models: z.array(comboModelEntry).min(1, "a combo requires at least one model"), strategy: comboStrategySchema.optional().default("priority"), config: comboRuntimeConfigSchema.optional(), allowedProviders: z.array(z.string().trim().min(1).max(200)).max(100).optional(), @@ -380,8 +380,9 @@ export const updateComboSchema = z .object({ name: comboNameSchema.optional(), description: z.string().max(2000).optional().nullable(), - // Creation may leave `models` empty (`omniroute combo create` drafts one - // that way); an update may not, or a working combo loses every target. + // An update may not remove every model from a combo, or a working combo + // loses every target. Creation refuses an empty list too: since the CLI + // gained --models (#10954), an empty draft has no remaining legitimate path. models: z .array(comboModelEntry) .min(1, "an update cannot remove every model from a combo") diff --git a/tests/unit/cli-combo-create-models-10954.test.ts b/tests/unit/cli-combo-create-models-10954.test.ts index dc53459d6ed..52fabf8c169 100644 --- a/tests/unit/cli-combo-create-models-10954.test.ts +++ b/tests/unit/cli-combo-create-models-10954.test.ts @@ -90,7 +90,15 @@ test("combo create — parses --models without throwing (Commander option regist }); await prog.parseAsync( - ["node", "x", "combo", "create", "my-combo", "--models", "openai/gpt-4o,anthropic/claude-3-opus"], + [ + "node", + "x", + "combo", + "create", + "my-combo", + "--models", + "openai/gpt-4o,anthropic/claude-3-opus", + ], { from: "node" } ); @@ -222,3 +230,26 @@ test("combo create (HTTP) — POST /api/combos body carries the parsed models", else process.env.DATA_DIR = ORIGINAL_DATA_DIR; } }); + +// Regression for the follow-up of #11011: with --models available, creating +// an empty combo is no longer a legitimate path on either transport. +test("combo create without any model is refused before reaching a transport", async () => { + await withComboEnv(async () => { + const errors: string[] = []; + const originalError = console.error; + console.error = (msg?: unknown) => { + errors.push(String(msg)); + }; + try { + const mod = await import("../../bin/cli/commands/combo.mjs"); + const rc = await mod.runComboCreateCommand("guard-test"); + assert.equal(rc, 1); + } finally { + console.error = originalError; + } + assert.ok( + errors.some((m) => m.includes("--models")), + `stderr should name --models, got: ${errors.join(" | ")}` + ); + }); +}); diff --git a/tests/unit/combo-bracket-names.test.ts b/tests/unit/combo-bracket-names.test.ts index 32844173cc1..0f4d8af0de9 100644 --- a/tests/unit/combo-bracket-names.test.ts +++ b/tests/unit/combo-bracket-names.test.ts @@ -30,6 +30,7 @@ test.after(() => { test("combo schemas accept names with spaces and square brackets", () => { const createResult = schemas.createComboSchema.safeParse({ name: "Claude [1m]", + models: ["anthropic/claude-3-opus"], }); const updateResult = schemas.updateComboSchema.safeParse({ name: "Claude [1m]", diff --git a/tests/unit/combo-context-length.test.ts b/tests/unit/combo-context-length.test.ts index f47539d64ad..c416b4d8be3 100644 --- a/tests/unit/combo-context-length.test.ts +++ b/tests/unit/combo-context-length.test.ts @@ -46,6 +46,7 @@ test.after(async () => { test("createComboSchema accepts valid context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 128000, }); assert.equal(result.success, true); @@ -54,6 +55,7 @@ test("createComboSchema accepts valid context_length", () => { test("createComboSchema rejects context_length below minimum (1000)", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 999, }); assert.equal(result.success, false); @@ -62,6 +64,7 @@ test("createComboSchema rejects context_length below minimum (1000)", () => { test("createComboSchema rejects context_length above maximum (2000000)", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 2000001, }); assert.equal(result.success, false); @@ -70,12 +73,14 @@ test("createComboSchema rejects context_length above maximum (2000000)", () => { test("createComboSchema accepts context_length at exact boundaries", () => { const min = schemas.createComboSchema.safeParse({ name: "MinCombo", + models: ["openai/gpt-4o-mini"], context_length: 1000, }); assert.equal(min.success, true); const max = schemas.createComboSchema.safeParse({ name: "MaxCombo", + models: ["openai/gpt-4o-mini"], context_length: 2000000, }); assert.equal(max.success, true); @@ -84,6 +89,7 @@ test("createComboSchema accepts context_length at exact boundaries", () => { test("createComboSchema rejects non-integer context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], context_length: 128000.5, }); assert.equal(result.success, false); @@ -92,6 +98,7 @@ test("createComboSchema rejects non-integer context_length", () => { test("createComboSchema accepts omitted context_length", () => { const result = schemas.createComboSchema.safeParse({ name: "TestCombo", + models: ["openai/gpt-4o-mini"], }); assert.equal(result.success, true); }); diff --git a/tests/unit/combo-empty-models.test.ts b/tests/unit/combo-empty-models.test.ts index 1486628ec53..95f32813a34 100644 --- a/tests/unit/combo-empty-models.test.ts +++ b/tests/unit/combo-empty-models.test.ts @@ -24,9 +24,9 @@ test("an update cannot remove every model from a combo", () => { assert.equal(updateComboSchema.safeParse({ name: "renamed" }).success, true); }); -test("creating a combo with no model stays allowed — the CLI does it on purpose", () => { - assert.equal(createComboSchema.safeParse({ name: "drafted", models: [] }).success, true); - assert.equal(createComboSchema.safeParse({ name: "drafted" }).success, true); +test("creating a combo without a model is refused at the boundary", () => { + assert.equal(createComboSchema.safeParse({ name: "drafted", models: [] }).success, false); + assert.equal(createComboSchema.safeParse({ name: "drafted" }).success, false); }); test("the copilot createCombo tool stores targets where the router looks for them", async () => { From 1058426120d4d22142901bb96b7f2f5bb83ab3b6 Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Sun, 23 Aug 2026 02:51:22 +0200 Subject: [PATCH 04/29] fix(executors): rotate on upstream 400 empty-body rejections (opencode) (#11158) Validated on the combined batch board over tip 92ef3c71: static gates clean (changelog, file-size, complexity 2624<=2774, cognitive 1182<=1223, dead-code 411<=416), typecheck:core clean, focused tests green. Empty-envelope 400 (no error field, empty content, finish_reason null) now rotates/retries instead of propagating as success; 200/streaming path never buffered; real-error 400s untouched. account-rotation + new rotation suite 34/34 on the board. Thank you @maxmad64bis! --- ...nding-opencode-empty-rejection-rotation.md | 1 + open-sse/executors/accountRotation.ts | 57 ++- open-sse/executors/opencode.ts | 70 ++- tests/unit/account-rotation.test.ts | 99 +++++ .../opencode-empty-rejection-rotation.test.ts | 416 ++++++++++++++++++ 5 files changed, 638 insertions(+), 5 deletions(-) create mode 100644 changelog.d/fixes/pending-opencode-empty-rejection-rotation.md create mode 100644 tests/unit/opencode-empty-rejection-rotation.test.ts diff --git a/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md new file mode 100644 index 00000000000..82a88c5905a --- /dev/null +++ b/changelog.d/fixes/pending-opencode-empty-rejection-rotation.md @@ -0,0 +1 @@ +- **fix(executors):** OpencodeExecutor rotates (or retries once on a single-account direct path) on upstream 400 empty-body rejections — malformed completion envelopes with no error field were propagated as success and killed client sessions. Bounded +1 attempt per request; body reads are conditioned on status 400 so successful/streaming responses are never buffered. 400s carrying an error field keep propagating immediately. diff --git a/open-sse/executors/accountRotation.ts b/open-sse/executors/accountRotation.ts index a64321e83d2..67115bdd8d0 100644 --- a/open-sse/executors/accountRotation.ts +++ b/open-sse/executors/accountRotation.ts @@ -1,7 +1,7 @@ /** * Shared multi-account rotation mechanics for noauth executors that round-robin * across several "accounts" (fingerprints), each with an optional dedicated - * proxy — currently `OpencodeExecutor` and `MimocodeExecutor`. + * proxy — currently `OpencodeExecutor`. * * Extracted after both executors independently implemented the same * pickAccount/markCooldown/markSuccess skeleton with the same exponential @@ -120,3 +120,58 @@ export function maskAccountId(fingerprint: string): string { export function isNetworkErrorRotatable(account: RotatableAccount): boolean { return account.proxy !== null; } + +/** + * Detect an *empty* upstream rejection: a 400 whose body carries no usable + * completion — the kind `OpencodeExecutor` must rotate/retry on instead of + * propagating as a fatal success. + * + * Signature is deliberately strict and scoped to the observed malformed + * envelope (`choices[0].message` with no `error`, no real `content`, + * `finish_reason: null`): + * - status must be exactly 400 (anything else → false); + * - body must parse and contain a `choices` array with at least one entry + * holding a `message` object; + * - an `error` field (present or empty) → false, so genuine 400s keep + * propagating immediately (#10460 precedent: classify by signature before + * rotating); + * - `tool_calls` / `reasoning_content` → false (real content); + * - `message.content` absent / null / "" → eligible; any other value + * (non-empty text, number, block array…) → false (conservative); + * - a literal `finish_reason` (not null) → false (a completed, if empty, turn). + * + * Does NOT reuse `detectMalformedNonStream` (diagnostics.ts): that classifier + * also flags `{error:{…}}` bodies as `empty_choices`, which would rotate on + * real errors — a false-positive class with a history here. + */ +export function isEmptyUpstreamRejection(status: number, bodyText: string): boolean { + if (status !== 400) return false; + let parsed: unknown; + try { + parsed = JSON.parse(bodyText); + } catch { + return false; + } + const choices = (parsed as { choices?: unknown })?.choices; + if (!Array.isArray(choices) || choices.length === 0) return false; + const first = choices[0] as { message?: unknown; finish_reason?: unknown }; + if (typeof first !== "object" || first === null) return false; + const rawMessage = (first as { message?: unknown }).message; + if (typeof rawMessage === "undefined" || rawMessage === null) return false; + if (typeof parsed !== "object" || parsed === null) return false; + if ("error" in (parsed as Record)) return false; + const msg = rawMessage as Record; + if ("tool_calls" in msg) return false; + if ("reasoning_content" in msg) return false; + const content = msg.content; + if (content !== undefined && content !== null && content !== "") return false; + if (first.finish_reason !== null && first.finish_reason !== undefined) return false; + return true; +} + +/** Best-effort extraction of the upstream `chatcmpl_*` id from a response body, + * for observability logging. Returns `"unknown"` when absent or unparseable. */ +export function extractChatcmplId(bodyText: string): string { + const match = /"id"\s*:\s*"(chatcmpl_[^"]+)"/.exec(bodyText); + return match ? match[1] : "unknown"; +} diff --git a/open-sse/executors/opencode.ts b/open-sse/executors/opencode.ts index c00ae258a3d..0829bfa871d 100644 --- a/open-sse/executors/opencode.ts +++ b/open-sse/executors/opencode.ts @@ -15,6 +15,8 @@ import { markSuccess as markAccountSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, } from "./accountRotation.ts"; import { isNetworkRotationSharedEgressGuardEnabled } from "@/shared/utils/featureFlags"; @@ -253,14 +255,41 @@ export class OpencodeExecutor extends BaseExecutor { try { this.syncAccountsFromCredentials(input.credentials); + const { log } = input; const hasProxies = this.accounts.some((a) => a.proxy !== null); - // Fast path: no multi-account proxy wiring configured → original behavior. + // Fast path: no multi-account proxy wiring configured → original behavior, + // plus exactly ONE bounded retry when the upstream answers a 400 empty + // rejection (same predicate and logging as the rotation loop). Everything + // else passes untouched: this path deliberately preserves BaseExecutor's + // intra-URL 429 retries (no skipUpstreamRetry here). if (this.accounts.length === 1 && !hasProxies) { - return await super.execute(input); + const single = (await super.execute(input)) as HttpExecuteResult; + if (single.response.status === 400) { + let bodyText: string | null = null; + try { + bodyText = await single.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on direct account"); + } + if (bodyText !== null) { + if (isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on direct account (${chatcmplId}), retrying once…` + ); + return await super.execute(input); + } + log?.debug?.( + "OPENCODE", + "400 without error field, signature not matched on direct account — observing" + ); + } + } + return single; } - const { log } = input; // This loop only ever dispatches through super.execute() (the HTTP request // path), which always resolves the object-shaped arm of ExecutorExecuteResult // — the bare-Response arm belongs to web/scraping executors only (base.ts:290). @@ -277,8 +306,13 @@ export class OpencodeExecutor extends BaseExecutor { // network call, but proxied accounts (independent egress) are still // tried normally. let sharedEgressDown = false; + // Bounded extra attempts for empty upstream rejections: +1 for a single + // account (retry the same one), none for a multi-account fleet (rotation + // through the accounts is the retry). Avoids an unbounded loop on a + // persistently malformed upstream. + const emptyRejectionBudget = this.accounts.length === 1 ? 1 : 0; - for (let attempt = 0; attempt < this.accounts.length; attempt++) { + for (let attempt = 0; attempt < this.accounts.length + emptyRejectionBudget; attempt++) { const account = this.pickAccount(); const masked = maskAccountId(account.fingerprint); @@ -354,6 +388,34 @@ export class OpencodeExecutor extends BaseExecutor { continue; } + // Empty upstream rejection (malformed 400: no error field, no real + // content, finish_reason null — see isEmptyUpstreamRejection). Rotate/ + // retry instead of propagating it as a fatal success: the observed + // envelope was marking subagent sessions as failed. Read the body ONLY + // for a 400 (never a 200/streaming — that would buffer the good path); + // classify, log, and continue. Neitheries markCooldown nor markSuccess: + // the failure is upstream's, not this account's. + if (status === 400) { + let bodyText: string | null = null; + try { + bodyText = await result.response.clone().text(); + } catch { + log?.debug?.("OPENCODE", "body read failed on empty rejection check"); + } + if (bodyText !== null && isEmptyUpstreamRejection(400, bodyText)) { + const chatcmplId = extractChatcmplId(bodyText); + log?.warn?.( + "OPENCODE", + `upstream empty rejection on account ${masked} (${chatcmplId}), rotating to next…` + ); + continue; + } + // A 400 carrying a real error (or non-empty content): propagate + // immediately, untouched — same as before this change. + this.markSuccess(account); + return result; + } + this.markSuccess(account); return result; } diff --git a/tests/unit/account-rotation.test.ts b/tests/unit/account-rotation.test.ts index f4864ee2dd8..a682f02a0d8 100644 --- a/tests/unit/account-rotation.test.ts +++ b/tests/unit/account-rotation.test.ts @@ -7,6 +7,8 @@ import { markSuccess, maskAccountId, isNetworkErrorRotatable, + isEmptyUpstreamRejection, + extractChatcmplId, type RotatableAccount, } from "../../open-sse/executors/accountRotation.ts"; @@ -114,3 +116,100 @@ describe("accountRotation", () => { assert.strictEqual(isNetworkErrorRotatable(withoutProxy), false); }); }); + +describe("isEmptyUpstreamRejection", () => { + it("matches the observed malformed completion envelope (no error field, empty content, null finish_reason)", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(isEmptyUpstreamRejection(400, observed), true); + }); + + it("does not match a non-400 status", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(isEmptyUpstreamRejection(200, observed), false); + assert.strictEqual(isEmptyUpstreamRejection(429, observed), false); + assert.strictEqual(isEmptyUpstreamRejection(502, observed), false); + }); + + it("does not match when an error field is present", () => { + const withError = JSON.stringify({ + error: { message: "bad request", type: "invalid_request_error" }, + }); + assert.strictEqual(isEmptyUpstreamRejection(400, withError), false); + const emptyError = JSON.stringify({ error: {} }); + assert.strictEqual(isEmptyUpstreamRejection(400, emptyError), false); + }); + + it("does not match when content is non-empty or tool_calls present", () => { + const nonEmpty = JSON.stringify({ + choices: [{ message: { role: "assistant", content: "hi" }, finish_reason: "stop" }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, nonEmpty), false); + + const toolCalls = JSON.stringify({ + choices: [ + { message: { role: "assistant", tool_calls: [{ id: "x" }] }, finish_reason: "tool_calls" }, + ], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, toolCalls), false); + }); + + it("does not match when content is a non-string non-null value (number, block array)", () => { + const numericContent = JSON.stringify({ + choices: [{ message: { role: "assistant", content: 123 }, finish_reason: null }], + }); + assert.strictEqual( + isEmptyUpstreamRejection(400, numericContent), + false, + "non-string non-null content is not eligible" + ); + + const reasoningContent = JSON.stringify({ + choices: [ + { message: { role: "assistant", reasoning_content: "thinking" }, finish_reason: null }, + ], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, reasoningContent), false); + }); + + it("does not match when choices or message are absent", () => { + const noChoices = JSON.stringify({ id: "chatcmpl_x", model: "muse" }); + assert.strictEqual(isEmptyUpstreamRejection(400, noChoices), false); + const noMessage = JSON.stringify({ choices: [{ finish_reason: null }] }); + assert.strictEqual(isEmptyUpstreamRejection(400, noMessage), false); + }); + + it("does not match when finish_reason is a literal value (not null)", () => { + const stopReason = JSON.stringify({ + choices: [{ message: { role: "assistant" }, finish_reason: "stop" }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, stopReason), false); + }); + + it("matches an empty string content (treated as eligible)", () => { + const emptyContent = JSON.stringify({ + choices: [{ message: { role: "assistant", content: "" }, finish_reason: null }], + }); + assert.strictEqual(isEmptyUpstreamRejection(400, emptyContent), true); + }); + + it("returns false for unparseable JSON rather than throwing", () => { + assert.strictEqual(isEmptyUpstreamRejection(400, "not json"), false); + assert.strictEqual(isEmptyUpstreamRejection(400, ""), false); + }); +}); + +describe("extractChatcmplId", () => { + it("extracts the chatcmpl id from an observed envelope", () => { + const observed = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; + assert.strictEqual(extractChatcmplId(observed), "chatcmpl_44fn2g6e7kk"); + }); + + it("falls back to 'unknown' when no id is present", () => { + assert.strictEqual(extractChatcmplId("{choices:[]}"), "unknown"); + assert.strictEqual(extractChatcmplId(""), "unknown"); + assert.strictEqual(extractChatcmplId("not json"), "unknown"); + }); +}); diff --git a/tests/unit/opencode-empty-rejection-rotation.test.ts b/tests/unit/opencode-empty-rejection-rotation.test.ts new file mode 100644 index 00000000000..784a4b83a74 --- /dev/null +++ b/tests/unit/opencode-empty-rejection-rotation.test.ts @@ -0,0 +1,416 @@ +import { describe, it, beforeEach, afterEach, before, after } from "node:test"; +import assert from "node:assert"; +import net from "node:net"; +import { OpencodeExecutor } from "../../open-sse/executors/opencode.ts"; +import type { ExecutorLog, ProviderCredentials } from "../../open-sse/executors/base.ts"; +import { resolveProxyForRequest } from "../../open-sse/utils/proxyFetch.ts"; +import { + isEmptyUpstreamRejection, + extractChatcmplId, +} from "../../open-sse/executors/accountRotation.ts"; + +/** + * Empty-upstream-rejection rotation (#design opencode-empty-rejection-rotation). + * + * An upstream 400 whose body carries no usable completion (the observed malformed + * envelope: `choices[0].message` with no error field, no real content, + * `finish_reason: null`) must be rotated/retried instead of propagated as a fatal + * success — that was killing subagent sessions. These tests pin the wiring: + * + * 1. A 400 empty rejection rotates to the next account (and its proxy). + * 2. The retry budget is bounded: +1 attempt for a single account, exactly N + * for an N-account all-empty run (propagate the last 400, never loop forever). + * 3. A 400 carrying a real error field (or non-empty content) still propagates + * immediately — no cooldown, no success, no rotation. + * 4. The 200/success path is never cloned or read (anti-bufferisation). + * + * The dispatch layer is mocked by stubbing globalThis.fetch (exactly what the + * #4954 proxy integration test does). Three throwaway TCP listeners stand in for + * the per-account proxies so runWithProxyContext's reachability probe passes. + */ + +const log: ExecutorLog = { debug() {}, info() {}, warn() {}, error() {} }; + +const ACCOUNT_A = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; +const ACCOUNT_B = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; +const ACCOUNT_C = "cccccccccccccccccccccccccccccccc"; + +const EMPTY_BODY = + '{"id":"chatcmpl_44fn2g6e7kk","object":"chat.completion","created":1787419957,"model":"muse-spark-1.2-contributor-free","choices":[{"index":0,"message":{"role":"assistant"},"finish_reason":null}]}'; +const ERROR_BODY = JSON.stringify({ + error: { message: "bad request", type: "invalid_request_error" }, +}); + +let serverA: net.Server; +let serverB: net.Server; +let serverC: net.Server; +let portA = 0; +let portB = 0; +let portC = 0; + +function listen(server: net.Server): Promise { + return new Promise((resolve) => { + server.listen(0, "127.0.0.1", () => { + resolve((server.address() as net.AddressInfo).port); + }); + }); +} + +before(async () => { + serverA = net.createServer((s) => s.destroy()); + serverB = net.createServer((s) => s.destroy()); + serverC = net.createServer((s) => s.destroy()); + portA = await listen(serverA); + portB = await listen(serverB); + portC = await listen(serverC); +}); + +after(() => { + serverA?.close(); + serverB?.close(); + serverC?.close(); +}); + +function portFor(fp: string): number { + if (fp === ACCOUNT_A) return portA; + if (fp === ACCOUNT_B) return portB; + return portC; +} + +/** `fingerprints` accounts; `proxied` is the subset that get a dedicated proxy + * (defaults to all). A proxy-less account shares the default egress. */ +function credentialsFor( + fingerprints: string[], + proxied: string[] = [...fingerprints] +): ProviderCredentials { + return { + apiKey: null, + accessToken: null, + connectionId: "noauth", + providerSpecificData: { + fingerprints, + ...(proxied.length > 0 && { + accountProxies: proxied.map((fp) => ({ + fingerprint: fp, + proxy: { type: "http", host: "127.0.0.1", port: portFor(fp) }, + })), + }), + }, + }; +} + +/** A Response subclass that counts clone() so we can assert the executor never + * buffers a 200/streaming response. Note: `clone()` returns a plain Response, so + * only `clone()` is reliably counted (a read on the clone hits the native + * method, not this override) — counting clones is the meaningful invariant. */ +class SpyResponse extends Response { + static clones = 0; + clone(): Response { + SpyResponse.clones++; + return super.clone(); + } +} + +interface PlanStep { + status: number; + body?: string; + throw?: Error; +} + +describe("OpencodeExecutor empty-rejection rotation", () => { + let originalFetch: typeof globalThis.fetch; + let observed: Array<{ source: string; host: string | null; port: string | null }>; + const GUARD_FLAG = "NETWORK_ROTATION_SHARED_EGRESS_GUARD"; + let savedGuardFlag: string | undefined; + + beforeEach(() => { + originalFetch = globalThis.fetch; + observed = []; + SpyResponse.clones = 0; + savedGuardFlag = process.env[GUARD_FLAG]; + delete process.env[GUARD_FLAG]; + }); + + afterEach(() => { + globalThis.fetch = originalFetch; + if (savedGuardFlag === undefined) delete process.env[GUARD_FLAG]; + else process.env[GUARD_FLAG] = savedGuardFlag; + }); + + function installFetch(plan: PlanStep[]) { + let call = 0; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = + typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + const resolved = resolveProxyForRequest(url); + observed.push({ + source: resolved.source, + host: resolved.proxyUrl ? new URL(resolved.proxyUrl).hostname : null, + port: resolved.proxyUrl ? new URL(resolved.proxyUrl).port : null, + }); + const step = plan[Math.min(call, plan.length - 1)]; + call++; + if (step.throw) throw step.throw; + return new SpyResponse(step.body ?? JSON.stringify({ ok: step.status === 200 }), { + status: step.status, + headers: { "Content-Type": "application/json" }, + }); + }) as typeof globalThis.fetch; + } + + /** + * Launches the executor. Asserts the predicate itself behaves (regression guard + * for the design's signature — the wiring tests below depend on it). + */ + it("predicate matches the observed envelope and rejects real errors", () => { + assert.strictEqual(isEmptyUpstreamRejection(400, EMPTY_BODY), true); + assert.strictEqual(isEmptyUpstreamRejection(200, EMPTY_BODY), false); + assert.strictEqual(isEmptyUpstreamRejection(400, ERROR_BODY), false); + assert.strictEqual(extractChatcmplId(EMPTY_BODY), "chatcmpl_44fn2g6e7kk"); + }); + + it("rotates to the next account on an empty 400 rejection (loop)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 200, + "must rotate past the empty 400" + ); + assert.ok(observed.length >= 2, "should have dispatched on a second account"); + assert.ok( + observed.some((o) => o.port === String(portA)), + "first attempt on account A" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "rotated attempt on account B" + ); + }); + + it("caps an all-empty N-account run at N attempts and propagates the last 400 intact", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + { status: 200 }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 400, + "must propagate the last empty 400" + ); + assert.strictEqual(observed.length, 3, "must NOT exceed N attempts (no infinite loop)"); + assert.ok(SpyResponse.clones >= 1, "the empty 400 path must read the body to classify it"); + const propagated = await (result as { response: Response }).response.clone().text(); + assert.strictEqual(propagated, EMPTY_BODY, "propagated 400 body must stay intact"); + for (const p of observed) { + assert.strictEqual(p.source, "context", "every dispatch must egress through a proxy context"); + } + }); + + it("retries the same proxied account once when it is the only account", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A]), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + assert.strictEqual(observed.length, 2, "exactly one bounded retry on the sole account"); + assert.ok( + observed.every((o) => o.port === String(portA)), + "both attempts egress through the single account's proxy" + ); + }); + + it("coexists with 429 rotation and 200 success in the same request", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 429 }, { status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 200, + "final response should succeed" + ); + assert.strictEqual(observed.length, 3, "429 + empty-400 + success across three accounts"); + assert.ok( + observed.some((o) => o.port === String(portA)), + "account A (429)" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "account B (empty 400)" + ); + assert.ok( + observed.some((o) => o.port === String(portC)), + "account C (200)" + ); + }); + + it("propagates a 400 carrying an error field immediately (no rotation)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: ERROR_BODY }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: Response }).response.status, + 400, + "real error 400 must propagate" + ); + assert.strictEqual(observed.length, 1, "must NOT rotate on a genuine error 400"); + }); + + it("never clones or reads the body of a 200 via the loop", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 200 }, { status: 200 }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B, ACCOUNT_C]), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 200); + assert.strictEqual(SpyResponse.clones, 0, "loop 200 must never be cloned"); + }); + + it("retries once via the fast path when a direct account answers an empty 400", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + assert.strictEqual(observed.length, 2, "fast path must retry the direct account exactly once"); + }); + + it("propagates the second 400 intact when the fast path retries and empty-rejects again", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([ + { status: 400, body: EMPTY_BODY }, + { status: 400, body: EMPTY_BODY }, + ]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 400); + const propagated = await (result as { response: Response }).response.clone().text(); + assert.strictEqual(propagated, EMPTY_BODY, "second rejection propagates with intact body"); + assert.strictEqual(observed.length, 2, "exactly one retry, no loop"); + }); + + it("never clones or reads the body of a 200 via the fast path", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + credentials: credentialsFor([ACCOUNT_A], []), + log, + }); + + assert.strictEqual((result as { response: Response }).response.status, 200); + assert.strictEqual(SpyResponse.clones, 0, "fast path 200 must never be cloned"); + }); + + it("rotates to a proxied account after a proxy-less account empty-rejects (shared-egress guard on by default)", async () => { + const exec = new OpencodeExecutor("opencode-zen"); + installFetch([{ status: 400, body: EMPTY_BODY }, { status: 200 }]); + + const result = await exec.execute({ + model: "deepseek-v4-flash-free", + body: { messages: [{ role: "user", content: "hi" }], stream: false }, + stream: false, + signal: null, + // A proxy-less, B proxied: B must still be tried and succeed. + credentials: credentialsFor([ACCOUNT_A, ACCOUNT_B], [ACCOUNT_B]), + log, + }); + + assert.strictEqual( + (result as { response: { status: number } }).response.status, + 200, + "the proxied account (B) must still be tried and must succeed" + ); + assert.strictEqual(observed.length, 2, "exactly one empty rejection (A) then one success (B)"); + assert.ok( + observed.some((o) => o.source === "direct"), + "first dispatch on the proxy-less account" + ); + assert.ok( + observed.some((o) => o.port === String(portB)), + "rotated dispatch on the proxied account" + ); + }); +}); From 60f25eff98b9c36270d07d03adee2bf3942f464c Mon Sep 17 00:00:00 2001 From: ggdayup Date: Sun, 23 Aug 2026 08:53:08 +0800 Subject: [PATCH 05/29] fix(sse): merge purify_history compression notice into the leading system message (#11113) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board: purify-system-first suite 4/4, typecheck clean. Pre-merge: file-size baseline gained a frozen entry for contextManager.ts at 1001 (+1, this PR's merge-into-leading-system branch) with a dated annotation — the gate caps unlisted files at 1000. Producer side of the live-confirmed TokenRouter 400 class: no internal path emits a mid-array system message anymore. Thank you @ggdayup — the call-log evidence made this airtight! --- config/quality/file-size-baseline.json | 2 + open-sse/services/contextManager.ts | 32 ++++++- ...ontext-manager-purify-system-first.test.ts | 91 +++++++++++++++++++ 3 files changed, 120 insertions(+), 5 deletions(-) create mode 100644 tests/unit/context-manager-purify-system-first.test.ts diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index b7f40cebfa6..f803789ffde 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -388,6 +388,8 @@ "open-sse/services/claudeCodeCompatible.ts": 1563, "open-sse/services/combo.ts": 4742, "open-sse/services/compression/strategySelector.ts": 1379, + "open-sse/services/contextManager.ts": 1001, + "_rebaseline_2026_08_22_11113_purify_system_first": "PR #11113 (ggdayup) own growth: open-sse/services/contextManager.ts 1000->1001 (+1, purifyHistory merges the compression notice into the leading system message instead of splicing a second one mid-array — live-confirmed TokenRouter 400s; the +1 is the merge-into-leading branch, not extractable). Covered by tests/unit/context-manager-purify-system-first.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "open-sse/services/rateLimitManager.ts": 1517, "open-sse/translator/response/openai-responses.ts": 1652, "open-sse/utils/cursorAgentProtobuf.ts": 1956, diff --git a/open-sse/services/contextManager.ts b/open-sse/services/contextManager.ts index a2d678f107c..6fe9e94c8e2 100644 --- a/open-sse/services/contextManager.ts +++ b/open-sse/services/contextManager.ts @@ -669,13 +669,35 @@ function purifyHistory(messages: Record[], targetTokens: number result = fixToolPairs(result); result = stripTrailingAssistantOrphanToolUse(result); - // Add summary of dropped messages + // Add summary of dropped messages. Merge the notice INTO the leading + // system/developer message instead of splicing a second system-role message + // mid-array: strict gateways (TokenRouter confirmed live 2026-08-22, see the + // PROVIDERS_SYSTEM_MUST_BE_FIRST list in src/lib/memory/injection.ts) reject + // any system message at index > 0 with HTTP 400 "System message must be at + // the beginning". When there is no leading system message, prepend one -- + // index 0 is accepted by every provider (same slot the old splice used when + // system[] was empty). if (keep < nonSystem.length) { const dropped = nonSystem.length - keep; - result.splice(system.length, 0, { - role: "system", - content: `[Context compressed: ${dropped} earlier messages removed to fit context window]`, - }); + const droppedNotice = `[Context compressed: ${dropped} earlier messages removed to fit context window]`; + const first = result[0]; + if (first && (first.role === "system" || first.role === "developer")) { + if (typeof first.content === "string") { + result[0] = { + ...first, + content: first.content ? `${droppedNotice}\n${first.content}` : droppedNotice, + }; + } else if (Array.isArray(first.content)) { + result[0] = { + ...first, + content: [{ type: "text", text: droppedNotice }, ...(first.content as unknown[])], + }; + } else { + result[0] = { ...first, content: droppedNotice }; + } + } else { + result.unshift({ role: "system", content: droppedNotice }); + } } return result; diff --git a/tests/unit/context-manager-purify-system-first.test.ts b/tests/unit/context-manager-purify-system-first.test.ts new file mode 100644 index 00000000000..51c658231d0 --- /dev/null +++ b/tests/unit/context-manager-purify-system-first.test.ts @@ -0,0 +1,91 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { compressContext } from "../../open-sse/services/contextManager.ts"; + +/** + * Plan-A root fix for the 2026-08-22 tokenrouter 400s. purifyHistory() used to + * splice the `[Context compressed: …]` notice as a SECOND system-role message at + * index system.length; strict gateways (TokenRouter, xiaomi-mimo/mimo) reject any + * system message at index > 0 with HTTP 400 "System message must be at the + * beginning". The notice must now merge into the leading system/developer + * message — or prepend a single system message when none exists — so the output + * never contains a system role after index 0, for ANY provider. + */ + +function bigTurn(n: number) { + return { role: "user", content: `turn ${n}: ${"x".repeat(4_000)}` }; +} + +function run(body: Record) { + // ~30k tokens of history vs a small target forces Layer-3 purify_history. + return compressContext(body, { maxTokens: 5_000, reserveTokens: 0 }); +} + +function systemIndices(messages: Array<{ role: string }>) { + return messages.map((m, i) => (m.role === "system" ? i : -1)).filter((i) => i >= 0); +} + +test("purify_history merges dropped-notice into existing leading system message", () => { + const body = { + model: "any-model", + messages: [ + { role: "system", content: "You are a helpful assistant." }, + ...Array.from({ length: 12 }, (_, i) => bigTurn(i)), + ], + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual(systemIndices(messages as Array<{ role: string }>).slice(1), []); + const first = messages[0]; + assert.equal(first.role, "system"); + const text = String(first.content); + assert.match(text, /Context compressed: \d+ earlier messages removed/); + assert.match(text, /You are a helpful assistant\./); +}); + +test("purify_history prepends a single system notice when no system message exists", () => { + const body = { + model: "any-model", + messages: Array.from({ length: 12 }, (_, i) => bigTurn(i)), + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual(systemIndices(messages as Array<{ role: string }>), [0]); + assert.match(String(messages[0].content), /Context compressed: \d+ earlier messages removed/); +}); + +test("purify_history merges into leading developer message without adding a second one", () => { + const body = { + model: "any-model", + messages: [ + { role: "developer", content: "dev instructions" }, + ...Array.from({ length: 12 }, (_, i) => bigTurn(i)), + ], + }; + const result = run(body); + assert.equal(result.compressed, true); + const messages = (result.body as { messages: Array> }).messages; + assert.deepEqual( + messages.filter((m) => m.role === "developer").length, + 1, + "exactly one developer message" + ); + assert.match(String(messages[0].content), /Context compressed: \d+ earlier messages removed/); + assert.match(String(messages[0].content), /dev instructions/); +}); + +test("no compression means no notice and untouched history", () => { + const body = { + model: "any-model", + messages: [ + { role: "system", content: "sys" }, + { role: "user", content: "hi" }, + ], + }; + const result = run(body); + assert.equal(result.compressed, false); + const messages = (result.body as { messages: unknown[] }).messages; + assert.equal(messages.length, 2); +}); From 80b8d2a8a284baff3e5740d0a086f1c68ff39531 Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Sun, 23 Aug 2026 02:54:48 +0200 Subject: [PATCH 06/29] fix(sse): resume stream recovery after a clean stop with reasoning-only output (#11151) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged after conflict resolution against the tip's #11109 (per-call tool_call tracking): scanOpenAiSseText keeps the per-call finish_reason special-case AND gains reasoningText + literal finishReason; canContinue uses the in-flight predicate with the new reasoning-only-clean-stop escape. One integration fix on the branch: the PR's hallucinatedEmptyStop referenced emittedToolCall, which #11109 had renamed — the branch now tracks emittedSawToolCall at the emitted level (any tool_call delta, complete or not), preserving the PR's don't-recover-after-tool-calls intent. Chain suites green: stream-continuation-wiring + stream-continuation + stream-recovery-toolcall 29/29. Thank you @maxmad64bis! --- open-sse/services/streamRecovery.ts | 105 +++++++++++++++--- tests/unit/stream-continuation-wiring.test.ts | 92 +++++++++++++++ tests/unit/stream-continuation.test.ts | 33 ++++++ 3 files changed, 214 insertions(+), 16 deletions(-) diff --git a/open-sse/services/streamRecovery.ts b/open-sse/services/streamRecovery.ts index 37a4867f137..129039c26cb 100644 --- a/open-sse/services/streamRecovery.ts +++ b/open-sse/services/streamRecovery.ts @@ -183,6 +183,11 @@ export function hasTerminalMarker(bytes: Uint8Array): boolean { export interface OpenAiSseScan { /** Concatenated assistant text seen across `choices[].delta.content`. */ text: string; + /** Concatenated reasoning trace seen across `choices[].delta.reasoning_content`. Some + * providers stream the entire answer here and leave `content` empty/null — tracked + * separately so a clean stop with reasoning-only output can still be recognized as + * "nothing usable was delivered" instead of "a normal empty turn". */ + reasoningText: string; /** True if any `choices[].delta.tool_calls` appeared — NEVER continue those. */ sawToolCall: boolean; /** @@ -201,6 +206,10 @@ export interface OpenAiSseScan { * (and the client-visible SSE) is still eligible to be resumed past it. */ terminal: boolean; + /** The literal `finish_reason` string when present (e.g. "stop", "tool_calls", "length", + * "content_filter"), or `null` if none was seen. `terminal` alone is not precise enough + * to gate the reasoning-only-stop continuation — it must fire on `"stop"` only. */ + finishReason: string | null; /** True if at least one OpenAI-shaped `choices[].delta` was parsed (format gate). */ parsedOpenAi: boolean; } @@ -212,12 +221,22 @@ export interface OpenAiSseScan { */ export function scanOpenAiSseText(sse: string): OpenAiSseScan { let text = ""; + let reasoningText = ""; let sawToolCall = false; let toolCallFinished = false; let terminal = false; + let finishReason: string | null = null; let parsedOpenAi = false; if (typeof sse !== "string" || sse.length === 0) { - return { text, sawToolCall, sawToolCallInFlight: false, terminal, parsedOpenAi }; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight: false, + terminal, + finishReason, + parsedOpenAi, + }; } for (const line of sse.split("\n")) { const trimmed = line.trimStart(); @@ -242,21 +261,33 @@ export function scanOpenAiSseText(sse: string): OpenAiSseScan { parsedOpenAi = true; const content = (delta as { content?: unknown }).content; if (typeof content === "string") text += content; + const reasoning = (delta as { reasoning_content?: unknown }).reasoning_content; + if (typeof reasoning === "string") reasoningText += reasoning; const toolCalls = (delta as { tool_calls?: unknown }).tool_calls; if (Array.isArray(toolCalls) && toolCalls.length > 0) sawToolCall = true; } - const finishReason = (choice as { finish_reason?: unknown })?.finish_reason; - if (finishReason === "tool_calls") { + const rawFinishReason = (choice as { finish_reason?: unknown })?.finish_reason; + if (rawFinishReason === "tool_calls") { // Ends this one choice, but the overall stream/turn stays continuable — // never counts as the general terminal marker (see OpenAiSseScan.terminal). toolCallFinished = true; - } else if (finishReason != null) { + finishReason = "tool_calls"; + } else if (rawFinishReason != null) { terminal = true; + if (typeof rawFinishReason === "string") finishReason = rawFinishReason; } } } const sawToolCallInFlight = sawToolCall && !toolCallFinished; - return { text, sawToolCall, sawToolCallInFlight, terminal, parsedOpenAi }; + return { + text, + reasoningText, + sawToolCall, + sawToolCallInFlight, + terminal, + finishReason, + parsedOpenAi, + }; } export interface ContinuableBody { @@ -267,8 +298,10 @@ export interface ContinuableBody { /** * Build a re-request body that continues from `assistantSoFar` by appending it as an - * assistant turn. Returns null when the body has no `messages` array or the partial text - * is empty (nothing to continue from). Does not mutate the original. + * assistant turn. When `assistantSoFar` is empty (nothing usable was emitted yet — e.g. a + * clean stop that only produced reasoning), the messages are re-sent unchanged instead of + * appending an empty assistant turn: this simply re-asks for a real answer. Returns null + * only when the body has no `messages` array at all (nothing to continue from). */ export function makeContinuationBody( body: ContinuableBody, @@ -276,10 +309,13 @@ export function makeContinuationBody( ): (ContinuableBody & { messages: unknown[] }) | null { if (!body || typeof body !== "object") return null; if (!Array.isArray(body.messages) || body.messages.length === 0) return null; - if (typeof assistantSoFar !== "string" || assistantSoFar.length === 0) return null; + if (typeof assistantSoFar !== "string") return null; return { ...body, - messages: [...body.messages, { role: "assistant", content: assistantSoFar }], + messages: + assistantSoFar.length > 0 + ? [...body.messages, { role: "assistant", content: assistantSoFar }] + : [...body.messages], stream: true, }; } @@ -390,8 +426,13 @@ export function createRecoverableStream( let continuations = 0; let emittedTail = ""; // raw SSE not yet scanned (awaiting an event boundary) let emittedText = ""; // assistant text already delivered to the client + let emittedReasoningText = ""; // reasoning trace already delivered (never shown to the client, + // tracked only to distinguish "a real empty turn" from "the whole + // answer stayed in the reasoning channel") + let emittedFinishReason: string | null = null; // literal finish_reason last seen, if any let emittedTerminal = false; let emittedToolCallInFlight = false; + let emittedSawToolCall = false; // any tool_call delta seen, complete or not let emittedParsedOpenAi = false; // Enqueue to the client and, when continuation is enabled, fold the chunk into the @@ -409,8 +450,11 @@ export function createRecoverableStream( emittedTail = emittedTail.slice(boundary + 2); const scan = scanOpenAiSseText(complete); emittedText += scan.text; + emittedReasoningText += scan.reasoningText; + if (scan.finishReason !== null) emittedFinishReason = scan.finishReason; if (scan.terminal) emittedTerminal = true; if (scan.sawToolCallInFlight) emittedToolCallInFlight = true; + if (scan.sawToolCall) emittedSawToolCall = true; if (scan.parsedOpenAi) emittedParsedOpenAi = true; }; @@ -418,15 +462,42 @@ export function createRecoverableStream( for (const chunk of holdback.flush()) emit(controller, chunk); }; - // A post-commit truncation is continuable only for a plain-text OpenAI-compatible - // stream that has not finished and has no tool call in flight. + // A post-commit truncation is continuable for a plain-text OpenAI-compatible stream that + // has no tool call in flight, AND either: + // - has not finished yet (the original #4131 truncation case), or + // - finished with a literal finish_reason of "stop" but delivered nothing usable while a + // non-empty reasoning trace shows the provider spent its whole turn "thinking" and never + // turned that into an answer (some providers put the entire response in + // reasoning_content and leave content empty). Gated on the LITERAL "stop" value, not the + // generic `terminal` flag — `terminal` also covers "length"/"content_filter"/a bare + // [DONE], which are out of scope for this specific recovery. + // + // Known consequence of the hallucinatedEmptyStop path (flagged in cross-review, accepted as + // inherent to tryContinue's existing design, not new to this fix): the original upstream's + // `finish_reason:"stop"` chunk was already forwarded to the client via `emit()`'s unconditional + // `controller.enqueue(chunk)` (streamRecovery.ts:381) BEFORE this scan ever runs — that is how + // `emittedFinishReason`/`emittedTerminal` get set in the first place. So the client sees an + // empty "stop" marker from the original turn, then — once the continuation succeeds — the real + // answer plus a SECOND `emitCleanTerminal` from `tryContinue`. This mirrors what already + // happens for the pre-existing truncation-continuation case (a truncated stream can likewise + // have partially delivered SSE framing before `tryContinue` appends more); it is not a new + // double-close of the underlying `ReadableStream` (`controller.close()` runs exactly once, + // after `tryContinue` returns). An SSE client that treats a bare `finish_reason:"stop"` as an + // unconditional end-of-turn (rather than waiting for `[DONE]`) may need updating separately — + // out of scope for this fix, which targets the observed opencode/OmniRoute pairing where the + // client kept the connection open. + const hallucinatedEmptyStop = () => + emittedFinishReason === "stop" && + !emittedSawToolCall && + emittedText.length === 0 && + emittedReasoningText.length > 0; + const canContinue = () => continueEnabled && continuations < maxContinuations && emittedParsedOpenAi && !emittedToolCallInFlight && - !emittedTerminal && - emittedText.length > 0; + (emittedText.length > 0 ? !emittedTerminal : hallucinatedEmptyStop()); const emitCleanTerminal = (controller: ReadableStreamDefaultController) => { controller.enqueue( @@ -527,9 +598,11 @@ export function createRecoverableStream( const { done, value } = result; if (done) { if (holdback.committed) { - // Graceful end after commit: if it lacks a terminal marker it is a silent - // truncation — try to continue; otherwise (clean finish) just close. - if (!emittedTerminal && (await tryContinue(controller))) { + // Graceful end after commit: try a mid-stream continuation whenever canContinue() + // says the stream is worth continuing (silent truncation, or a clean-but-empty + // reasoning-only stop) — canContinue() is the single source of truth here, same as + // the read-error branch above. + if (await tryContinue(controller)) { runFinalize(); controller.close(); return; diff --git a/tests/unit/stream-continuation-wiring.test.ts b/tests/unit/stream-continuation-wiring.test.ts index 2f246e325d4..1081d6ea9ba 100644 --- a/tests/unit/stream-continuation-wiring.test.ts +++ b/tests/unit/stream-continuation-wiring.test.ts @@ -46,6 +46,10 @@ async function collectText(stream: ReadableStream): Promise const ROLE = 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n'; const content = (s: string) => `data: {"choices":[{"delta":{"content":${JSON.stringify(s)}}}]}\n\n`; +const reasoning = (s: string) => + `data: {"choices":[{"delta":{"reasoning_content":${JSON.stringify(s)}}}]}\n\n`; +const finishStopNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'; +const finishLengthNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"length"}]}\n\n'; test("mid-stream continuation: stitches the suffix after a silent post-commit truncation", async () => { // Commits on chunk 1, emits "Hello wor", then ends WITHOUT a terminal marker (silent cut). @@ -115,3 +119,91 @@ test("tool-call in flight is never continued (would corrupt tool JSON)", async ( await collectText(stream); assert.equal(continued, false, "continuation must NOT fire once a tool call has started streaming"); }); + +test("mid-stream continuation: a clean stop with reasoning-only output (no answer) triggers a continuation", async () => { + const initial = streamFrom([ + ROLE, + reasoning("the model thinks through the problem here..."), + finishStopNoContent, + ]); + let continueArg = "__unset__"; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async (soFar: string) => { + continueArg = soFar; + return streamFrom([content("Here is the actual answer."), "data: [DONE]\n\n"]); + }, + }); + const out = await collectText(stream); + const scan = scanOpenAiSseText(out); + assert.equal(continueArg, "", "nothing usable was emitted — the re-request has an empty prefill"); + assert.equal( + scan.text, + "Here is the actual answer.", + "the client gets a real answer instead of silence" + ); + assert.equal(scan.terminal, true); +}); + +test("mid-stream continuation: a clean stop with truly empty output (no text, no reasoning) is left unchanged", async () => { + const initial = streamFrom([ROLE, finishStopNoContent]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + await collectText(stream); + assert.equal( + continued, + false, + "no reasoning trace means there is nothing to act on — do not guess" + ); +}); + +test("mid-stream continuation: finish_reason 'length' with reasoning-only output does NOT trigger a continuation", async () => { + // Regression guard for a blocker found in cross-review: widening the gate to any + // terminal marker (instead of the literal finish_reason "stop") would wrongly spend a + // continuation attempt on a token-limit cutoff, which is out of this fix's scope. + const initial = streamFrom([ + ROLE, + reasoning("the model was still thinking when it hit the token limit..."), + finishLengthNoContent, + ]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + await collectText(stream); + assert.equal(continued, false, "finish_reason 'length' is out of scope for this fix"); +}); + +test("mid-stream continuation: real content alongside reasoning at a clean stop is left unchanged (non-regression)", async () => { + const initial = streamFrom([ + ROLE, + reasoning("thinking..."), + content("The real answer."), + finishStopNoContent, + ]); + let continued = false; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => { + continued = true; + return streamFrom([content("nope"), "data: [DONE]\n\n"]); + }, + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal(continued, false, "real content was delivered — nothing to recover"); + assert.equal(scan.text, "The real answer."); +}); diff --git a/tests/unit/stream-continuation.test.ts b/tests/unit/stream-continuation.test.ts index caffe936df9..8a1da37ce9d 100644 --- a/tests/unit/stream-continuation.test.ts +++ b/tests/unit/stream-continuation.test.ts @@ -21,6 +21,30 @@ test("scanOpenAiSseText accumulates content deltas and flags an OpenAI-compat st assert.equal(r.terminal, false); }); +test("scanOpenAiSseText accumulates reasoning_content deltas separately from content", () => { + const sse = + 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n' + + 'data: {"choices":[{"delta":{"reasoning_content":"thinking..."}}]}\n\n' + + 'data: {"choices":[{"delta":{"reasoning_content":" more"}}]}\n\n'; + const r = scanOpenAiSseText(sse); + assert.equal(r.reasoningText, "thinking... more"); + assert.equal(r.text, "", "reasoning_content must never leak into the visible text field"); + assert.equal(r.parsedOpenAi, true); +}); + +test("scanOpenAiSseText captures the literal finish_reason value", () => { + const stop = scanOpenAiSseText('data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'); + assert.equal(stop.finishReason, "stop"); + + const length = scanOpenAiSseText( + 'data: {"choices":[{"delta":{"content":"x"},"finish_reason":"length"}]}\n\n' + ); + assert.equal(length.finishReason, "length"); + + const none = scanOpenAiSseText('data: {"choices":[{"delta":{"content":"x"}}]}\n\n'); + assert.equal(none.finishReason, null, "no finish_reason seen means null, not a guessed default"); +}); + test("scanOpenAiSseText detects the terminal [DONE] marker", () => { const r = scanOpenAiSseText('data: {"choices":[{"delta":{"content":"hi"}}]}\n\ndata: [DONE]\n\n'); assert.equal(r.text, "hi"); @@ -65,6 +89,15 @@ test("makeContinuationBody refuses bodies without a messages array or empty text assert.equal(makeContinuationBody(null as never, "t"), null); }); +test("makeContinuationBody accepts an empty prefill by re-sending the messages unchanged", () => { + const body = { model: "x", stream: true, messages: [{ role: "user", content: "hi" }] }; + const out = makeContinuationBody(body, ""); + assert.ok(out, "an empty prefill must still produce a re-request body, not null"); + assert.equal(out!.messages.length, 1, "no empty assistant turn is appended"); + assert.deepEqual(out!.messages[0], { role: "user", content: "hi" }); + assert.equal(out!.stream, true); +}); + // ── trimContinuationOverlap ─────────────────────────────────────────────────── test("trimContinuationOverlap removes a duplicated seam so the join is append-only", () => { From 62ab93d789c591c2d10889d62fe80f63058fa722 Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Sun, 23 Aug 2026 02:56:26 +0200 Subject: [PATCH 07/29] fix(sse): reject a low-overlap stream-recovery continuation instead of concatenating it raw (#11152) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged after sibling #11151 landed: streamRecovery.ts auto-merged byte-identical to the validated combined board; the test-file conflict (both PRs added suites at the same anchor) resolved keeping all 11 tests — #11151's four clean-stop cases plus this PR's three threshold cases, with the PR's updated partial-tail fixture for the pre-existing overlap test. Full chain green: 32/32 (wiring + continuation + toolcall regression). The documented 8-char overlap threshold ends the silent mid-word gluing. Thank you @maxmad64bis! --- open-sse/config/constants.ts | 17 ++++ open-sse/services/streamRecovery.ts | 19 +++- tests/unit/stream-continuation-wiring.test.ts | 99 +++++++++++++++++-- 3 files changed, 127 insertions(+), 8 deletions(-) diff --git a/open-sse/config/constants.ts b/open-sse/config/constants.ts index f1fa055c938..d99da4725b0 100644 --- a/open-sse/config/constants.ts +++ b/open-sse/config/constants.ts @@ -355,6 +355,23 @@ export const STREAM_RECOVERY = { HOLDBACK_MS: 750, BUFFER_MAX_BYTES: 65536, EARLY_RETRY_MAX: 4, + /** + * Minimum character overlap `trimContinuationOverlap` must find between the + * already-emitted text and a mid-stream continuation for the continuation to be + * accepted as a real resume, rather than an unrelated restart the model produced after + * ignoring the assistant-prefill. + * + * This is a DOCUMENTED TRADE-OFF, not a solved distinction: a model that continues + * cleanly with fewer than this many echoed characters (a legitimate, even preferred, + * outcome — there was nothing to de-duplicate) is indistinguishable, from string data + * alone, from a model that silently restarted on an unrelated sentence. Both produce a + * low/zero overlap. Rejecting below this threshold trades some false-positive rejections + * of legitimate low-overlap continuations (bounded retry, then a clean close — no data + * loss beyond that retry) against not silently gluing two unrelated fragments into one + * corrupted, unrecoverable answer. It does not eliminate the residual false negative + * either (an accidental coincidence at or above this many characters is still accepted). + */ + MIN_CONTINUATION_OVERLAP_CHARS: 8, } as const; /** diff --git a/open-sse/services/streamRecovery.ts b/open-sse/services/streamRecovery.ts index 129039c26cb..3a95e9a2a3f 100644 --- a/open-sse/services/streamRecovery.ts +++ b/open-sse/services/streamRecovery.ts @@ -541,7 +541,24 @@ export function createRecoverableStream( } const scan = scanOpenAiSseText(raw); - const suffix = trimContinuationOverlap(emittedText, scan.text); + // A continuation whose overlap with what was already emitted falls below the documented + // threshold is treated as a suspected restart rather than a real resume — see + // STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS for the full trade-off rationale. This + // is a heuristic, not a proof: it deliberately trades some false-positive rejections of + // legitimate low-overlap continuations against never silently gluing two unrelated + // fragments into one corrupted message. + const overlapResult = trimContinuationOverlap(emittedText, scan.text); + const overlapChars = scan.text.length - overlapResult.length; + const isSuspectedRestart = + emittedText.length > 0 && + scan.text.length > 0 && + overlapChars < STREAM_RECOVERY.MIN_CONTINUATION_OVERLAP_CHARS; + if (isSuspectedRestart) { + if (await tryContinue(controller)) return true; + emitCleanTerminal(controller); + return true; + } + const suffix = overlapResult; if (suffix) { emit( controller, diff --git a/tests/unit/stream-continuation-wiring.test.ts b/tests/unit/stream-continuation-wiring.test.ts index 1081d6ea9ba..36a6f445d1f 100644 --- a/tests/unit/stream-continuation-wiring.test.ts +++ b/tests/unit/stream-continuation-wiring.test.ts @@ -46,14 +46,16 @@ async function collectText(stream: ReadableStream): Promise const ROLE = 'data: {"choices":[{"delta":{"role":"assistant"}}]}\n\n'; const content = (s: string) => `data: {"choices":[{"delta":{"content":${JSON.stringify(s)}}}]}\n\n`; + const reasoning = (s: string) => `data: {"choices":[{"delta":{"reasoning_content":${JSON.stringify(s)}}}]}\n\n`; const finishStopNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n'; const finishLengthNoContent = 'data: {"choices":[{"delta":{},"finish_reason":"length"}]}\n\n'; test("mid-stream continuation: stitches the suffix after a silent post-commit truncation", async () => { - // Commits on chunk 1, emits "Hello wor", then ends WITHOUT a terminal marker (silent cut). - const initial = streamFrom([ROLE, content("Hello wor")]); + // Commits on chunk 1, emits "Hello there world", then ends WITHOUT a terminal marker + // (silent cut). + const initial = streamFrom([ROLE, content("Hello there world")]); let finalizeCount = 0; let continueArg = ""; @@ -64,15 +66,22 @@ test("mid-stream continuation: stitches the suffix after a silent post-commit tr now: steppingClock(), continueStream: async (soFar: string) => { continueArg = soFar; - // The model re-emits a small overlap ("wor") which must be trimmed away. - return streamFrom([ROLE, content("world!"), "data: [DONE]\n\n"]); + // The model re-emits only a partial tail of what was already sent ("there world", + // 11 chars — above the 8-char threshold, but NOT the full emitted text, unlike a + // full-string overlap this stays a discriminating test of trimContinuationOverlap's + // partial-tail trim, not just its "accept everything" path) before continuing. + return streamFrom([ROLE, content("there world, nice to meet you!"), "data: [DONE]\n\n"]); }, }); const out = await collectText(stream); const scan = scanOpenAiSseText(out); - assert.equal(continueArg, "Hello wor", "continuation is prefilled with the text already sent"); - assert.equal(scan.text, "Hello world!", "client sees the full answer, overlap trimmed, exactly once"); + assert.equal(continueArg, "Hello there world", "continuation is prefilled with the text already sent"); + assert.equal( + scan.text, + "Hello there world, nice to meet you!", + "client sees the full answer, partial overlap trimmed, exactly once" + ); assert.equal(scan.terminal, true, "the recovered stream ends with a terminal marker"); assert.equal(finalizeCount, 1, "finalize runs exactly once"); }); @@ -84,7 +93,7 @@ test("mid-stream continuation: recovers a post-commit transport error too", asyn const stream = createRecoverableStream(initial, async () => null, { finalize: () => {}, now: steppingClock(), - continueStream: async () => streamFrom([content("answer done."), "data: [DONE]\n\n"]), + continueStream: async () => streamFrom([content("Partial answer done."), "data: [DONE]\n\n"]), }); const scan = scanOpenAiSseText(await collectText(stream)); assert.equal(scan.text, "Partial answer done."); @@ -120,6 +129,82 @@ test("tool-call in flight is never continued (would corrupt tool JSON)", async ( assert.equal(continued, false, "continuation must NOT fire once a tool call has started streaming"); }); +test("mid-stream continuation: a zero-overlap restart is rejected, never concatenated raw", async () => { + // Truncates silently after real, non-empty text — canContinue() fires. + const initial = streamFrom([ROLE, content("Tous les faits sont reunis")]); + let continuations = 0; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + maxContinuations: 1, + continueStream: async () => { + continuations += 1; + // The model ignores the assistant prefill and restarts on an unrelated sentence — + // zero characters of overlap with what was already emitted. + return streamFrom([ + content("Je complete le design - derniere verification"), + "data: [DONE]\n\n", + ]); + }, + }); + const out = await collectText(stream); + const scan = scanOpenAiSseText(out); + assert.equal( + scan.text, + "Tous les faits sont reunis", + "the unrelated restart must never be appended to the already-emitted text" + ); + assert.equal(scan.terminal, true, "closes cleanly instead of leaving the client hanging"); + assert.equal(continuations, 1, "bounded by maxContinuations — does not loop forever"); +}); + +test("mid-stream continuation: a nonzero overlap below the threshold is rejected too", async () => { + // Genuine 4-character overlap ("pret"), well under the 8-char threshold — this is the + // false-negative case a naive `overlapChars === 0` check would miss (a restart that + // happens to share a short accidental fragment with the emitted tail): must still be + // treated as a suspected restart, not accepted as a genuine resume. + const initial = streamFrom([ROLE, content("Le design est pret")]); + let continuations = 0; + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + maxContinuations: 1, + continueStream: async () => { + continuations += 1; + // Shares only "pret" (4 chars) with the emitted tail, then diverges completely. + return streamFrom([content("pret a partir de zero"), "data: [DONE]\n\n"]); + }, + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal( + scan.text, + "Le design est pret", + "a below-threshold (but nonzero) overlap must not be accepted as a real resume" + ); + assert.equal(continuations, 1); +}); + +test("mid-stream continuation: a real overlap at or above the threshold is still stitched correctly", async () => { + // Regression guard: the existing happy path (first test in this file, whose updated + // fixture re-emits the 11-char partial tail "there world") still passes below — this test + // adds an overlap AT the threshold boundary to prove Task 3's new check does not fire when + // it shouldn't. + const initial = streamFrom([ROLE, content("The answer to this question")]); + const stream = createRecoverableStream(initial, async () => null, { + finalize: () => {}, + now: steppingClock(), + continueStream: async () => + // "question" (8 chars) overlaps the tail of emittedText exactly at the threshold. + streamFrom([content("question is forty-two."), "data: [DONE]\n\n"]), + }); + const scan = scanOpenAiSseText(await collectText(stream)); + assert.equal( + scan.text, + "The answer to this question is forty-two.", + "an overlap meeting the threshold is trimmed and stitched, not rejected" + ); +}); + test("mid-stream continuation: a clean stop with reasoning-only output (no answer) triggers a continuation", async () => { const initial = streamFrom([ ROLE, From eb9fa33ee736d500c84063ea7089f043761e306d Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 22 Aug 2026 22:23:17 -0300 Subject: [PATCH 08/29] feat(api): structured ?format=json for the self-service usage endpoint (#11190) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(api): structured ?format=json for the self-service usage endpoint GET /api/usage/om-usage already let any key read its own usage — personal daily/weekly USD limits and the provider quota snapshot — but only as text/plain, which a UI cannot parse safely. OmniCopilot issue #8 asks exactly for this surface. Adds ?format=json, returning the ApiKeyUsageLimitStatus + UsageSnapshot the text is rendered from. Text and JSON share the same collectors (collectUsageSnapshots, getApiKeyUsageLimitStatus), so the two can never disagree about a number. The response is a discriminated union: a key without allowUsageCommand (403) or an invalid key (401) returns { allowed:false, error:{message} }, distinct from allowed:true with empty sections — the state a panel must render as "nothing learned yet", not a refusal. Text form unchanged; without ?format the contract is untouched. The endpoint was previously missing from API_REFERENCE.md; it now has a section documenting both forms, the allowUsageCommand gate, and the self-service auth model (caller's own key, not requireManagementAuth). Regression guards in tests/unit/usage-command-json-format.test.ts (4 tests: json shape, text default preserved, structured 403, sanitized 401 with no stack trace). Existing internal-usage-command suite still 12/12. * chore(changelog): correct the fragment to the real PR number (#11190) --------- Co-authored-by: Xiangzhe --- .../features/11190-usage-command-json.md | 1 + docs/reference/API_REFERENCE.md | 41 ++++++ src/lib/usage/internalUsageCommand.ts | 90 +++++++++++- tests/unit/usage-command-json-format.test.ts | 130 ++++++++++++++++++ 4 files changed, 257 insertions(+), 5 deletions(-) create mode 100644 changelog.d/features/11190-usage-command-json.md create mode 100644 tests/unit/usage-command-json-format.test.ts diff --git a/changelog.d/features/11190-usage-command-json.md b/changelog.d/features/11190-usage-command-json.md new file mode 100644 index 00000000000..d7655f04c51 --- /dev/null +++ b/changelog.d/features/11190-usage-command-json.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage` gains a structured form — `?format=json` returns the key's own usage as `ApiKeyUsageLimitStatus` + `UsageSnapshot` instead of `text/plain`. This is the surface a UI (the OmniCopilot panel) consumes to show a key holder their daily/weekly spend and quota reset. The route is self-service (the caller's own key, gated by `allowUsageCommand`), not the management surface; refusals come back as a discriminated `{ "allowed": false, "error": … }` so a UI can tell "not allowed" apart from "allowed but nothing cached yet". The endpoint was previously undocumented in `API_REFERENCE.md`; it now has a section ([#11190](https://github.com/diegosouzapw/OmniRoute/pull/11190)) diff --git a/docs/reference/API_REFERENCE.md b/docs/reference/API_REFERENCE.md index 359ae4750ec..3e39f35ebaf 100644 --- a/docs/reference/API_REFERENCE.md +++ b/docs/reference/API_REFERENCE.md @@ -636,6 +636,47 @@ completion. --- +## Self-service usage (`/api/usage/om-usage`) + +Any API key can read **its own** usage and quotas — no management auth. This is the endpoint a +client (CLI, the OmniCopilot panel) uses to show a key holder their spend. + +```bash +# Text form (the historical contract — plain text for a terminal) +curl -H "Authorization: Bearer " \ + http://localhost:20128/api/usage/om-usage + +# Structured form — what a UI consumes +curl -H "Authorization: Bearer " \ + "http://localhost:20128/api/usage/om-usage?format=json" +``` + +The key must have **`allowUsageCommand`** enabled (off by default — the dashboard's API-key +manager toggles it per key). Without it the endpoint answers `403`. + +`?format=json` returns a discriminated shape so a caller never reads a data field off a +refusal. On success: + +```jsonc +{ + "allowed": true, + // present only when the key opted into per-key usage limits (daily/weekly USD): + "personal": { "dailySpentUsd": 1.25, "dailyLimitUsd": 5, "dailyResetAtIso": "…", "weeklySpentUsd": 8, "weeklyLimitUsd": 20, "weeklyResetAtIso": "…" /* … */ }, + // the selected provider quota snapshot, or null when nothing is cached yet: + "provider": { "connectionId": "…", "provider": "claude", "plan": "…", "quotas": { /* … */ } } +} +``` + +On refusal (`401` bad key / `403` not allowed) the same route returns +`{ "allowed": false, "error": { "message": "…" } }` — a present-but-empty `personal`/`provider` +(key allowed, nothing learned yet) is a different state from a refusal, and only the JSON form +distinguishes them. + +**Auth:** the caller's own Bearer API key, validated with `isValidApiKey` — this is *not* the +management surface (`/api/keys/…`), which stays behind `requireManagementAuth`. + +--- + ## Semantic Cache ```bash diff --git a/src/lib/usage/internalUsageCommand.ts b/src/lib/usage/internalUsageCommand.ts index 257b8f9c158..76d55315d9c 100644 --- a/src/lib/usage/internalUsageCommand.ts +++ b/src/lib/usage/internalUsageCommand.ts @@ -13,7 +13,7 @@ const TEXT_PLAIN_HEADERS = { "Content-Type": "text/plain; charset=utf-8" } as co type JsonRecord = Record; -interface UsageCommandApiKeyMetadata { +export interface UsageCommandApiKeyMetadata { id: string; name?: string; allowedConnections?: string[] | null; @@ -31,7 +31,7 @@ interface ProviderConnectionLike { quotaWindowThresholds?: Record | null; } -interface UsageSnapshot { +export interface UsageSnapshot { connectionId: string; provider: string; plan: unknown; @@ -39,7 +39,7 @@ interface UsageSnapshot { quotaWindowThresholds?: Record | null; } -interface UsageCommandSelection { +export interface UsageCommandSelection { preferredProvider?: string | null; preferredConnectionId?: string | null; } @@ -258,7 +258,7 @@ function snapshotFromConnection( }; } -async function collectUsageSnapshots( +export async function collectUsageSnapshots( metadata: UsageCommandApiKeyMetadata, deps: RequiredDeps ): Promise { @@ -525,6 +525,52 @@ function appendQuotaBlock( lines.push(`⏱ reset in ${formatResetIn(getResetAt(match?.quota ?? null), now)}`); } +/** + * Structured form of the usage command — what {@link buildUsageCommandText} + * renders as text, exposed as data for API consumers (the OmniCopilot panel + * asks for it via `?format=json`). Text and JSON share the exact same + * collectors, so the two can never disagree about a number. + * + * The key design constraint is the 403 case: a key without `allowUsageCommand` + * must reach the client as a *structured* reason, not a bare text error — a + * caller rendering a usage panel has to be able to tell "the server does not + * know your limits yet" apart from "this key may not ask". + */ +/** Discriminated so the caller never reads a data field off a refusal: + * `allowed:false` carries only `error`; `allowed:true` carries the data. */ +export type UsageCommandJson = + | { allowed: false; error: { message: string } } + | { + allowed: true; + /** Present only when the key opted into per-key usage limits. */ + personal: unknown | null; + /** The selected provider snapshot, or null when nothing is cached. */ + provider: UsageSnapshot | null; + }; + +export async function buildUsageCommandJson( + metadata: UsageCommandApiKeyMetadata, + deps: InternalUsageCommandDeps = {}, + selection: UsageCommandSelection = {} +): Promise { + const resolvedDeps = await normalizeDeps(deps); + const personal = + metadata.usageLimitEnabled === true + ? await resolvedDeps.getApiKeyUsageLimitStatus( + { + ...metadata, + preferredProvider: selection.preferredProvider ?? metadata.preferredProvider ?? null, + }, + { now: resolvedDeps.now } + ) + : null; + const provider = selectUsageSnapshot( + await collectUsageSnapshots(metadata, resolvedDeps), + selection + ); + return { allowed: true, personal, provider }; +} + export async function buildUsageCommandText( metadata: UsageCommandApiKeyMetadata, deps: InternalUsageCommandDeps = {}, @@ -588,6 +634,17 @@ function inferHttpUsageCommandSelection(request: Request): UsageCommandSelection } } +/** `?format=json` (or `?format=JSON`) — anything else falls back to the text + * form, which is the historical contract of this endpoint. */ +function wantsUsageCommandJson(request: Request): boolean { + try { + const format = new URL(request.url, "http://localhost").searchParams.get("format"); + return format !== null && format.trim().toLowerCase() === "json"; + } catch { + return false; + } +} + function createPlainUsageCommandResponse(text: string, status = 200): Response { return new Response(text, { status, headers: TEXT_PLAIN_HEADERS }); } @@ -764,22 +821,45 @@ export async function handleInternalUsageCommandHttpRequest( ): Promise { try { const resolvedDeps = await normalizeDeps(deps); + const json = wantsUsageCommandJson(request); const apiKey = extractUsageCommandApiKey(request); if (!apiKey || !(await resolvedDeps.isValidApiKey(apiKey))) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } const metadata = await resolvedDeps.getApiKeyMetadata(apiKey); if (!metadata?.id) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_AUTH_REQUIRED_MESSAGE } } satisfies UsageCommandJson, + { status: 401 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_AUTH_REQUIRED_MESSAGE, 401); } if (metadata.allowUsageCommand !== true) { + if (json) { + return Response.json( + { allowed: false, error: { message: USAGE_COMMAND_DISABLED_MESSAGE } } satisfies UsageCommandJson, + { status: 403 } + ); + } return createPlainUsageCommandResponse(USAGE_COMMAND_DISABLED_MESSAGE, 403); } + const selection = inferHttpUsageCommandSelection(request); + if (json) { + return Response.json(await buildUsageCommandJson(metadata, resolvedDeps, selection)); + } return createPlainUsageCommandResponse( - await buildUsageCommandText(metadata, resolvedDeps, inferHttpUsageCommandSelection(request)) + await buildUsageCommandText(metadata, resolvedDeps, selection) ); } catch (err) { const body = buildErrorBody(500, err instanceof Error ? err.message : String(err)); diff --git a/tests/unit/usage-command-json-format.test.ts b/tests/unit/usage-command-json-format.test.ts new file mode 100644 index 00000000000..f7d7b286c41 --- /dev/null +++ b/tests/unit/usage-command-json-format.test.ts @@ -0,0 +1,130 @@ +/** + * #8 (OmniCopilot) — the usage command answered `text/plain`, which a UI cannot + * parse safely. The structured form (`?format=json`) returns the same + * `ApiKeyUsageLimitStatus` + `UsageSnapshot` the text is rendered from. + * + * These tests pin the contract the extension depends on: JSON when asked, + * text by default, the 403 as a structured reason rather than a bare string, + * and an error body that never carries a stack trace. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { handleInternalUsageCommandHttpRequest } from "../../src/lib/usage/internalUsageCommand"; + +const NOW = Date.parse("2026-08-19T12:00:00.000Z"); + +const LIMIT_STATUS = { + enabled: true, + dailyLimitUsd: 5, + weeklyLimitUsd: 20, + dailySpentUsd: 1.25, + weeklySpentUsd: 8, + dailyWindowStartIso: "2026-08-19T03:00:00.000Z", + dailyResetAtIso: "2026-08-20T03:00:00.000Z", + weeklyWindowStartIso: "2026-08-16T03:00:00.000Z", + weeklyResetAtIso: "2026-08-23T03:00:00.000Z", + dailyExceeded: false, + weeklyExceeded: false, +}; + +function allowedDeps(overrides: Record = {}) { + return { + now: () => NOW, + isValidApiKey: async (apiKey: string) => apiKey === "sk-allowed", + getApiKeyMetadata: async () => ({ + id: "key-allowed", + name: "panel key", + allowUsageCommand: true, + usageLimitEnabled: true, + }), + getProviderConnections: async () => [ + { id: "conn-claude", provider: "claude", isActive: true }, + ], + getAllProviderLimitsCache: () => ({ + "conn-claude": { + plan: "Claude Max", + quotas: { + weekly: { used: 25, total: 100, remaining: 75, resetAt: "2026-08-25T03:00:00.000Z" }, + }, + message: null, + fetchedAt: new Date(NOW).toISOString(), + }, + }), + getProviderConnectionById: async () => null, + getProviderLimitsCache: () => null, + getQuotaPolicy: async () => ({ defaultThresholdPercent: 0, providerWindowDefaults: {} }), + getApiKeyUsageLimitStatus: async () => LIMIT_STATUS, + ...overrides, + }; +} + +test("om-usage ?format=json returns the structured personal + provider quota", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + assert.match(response.headers.get("content-type") ?? "", /application\/json/); + const body = (await response.json()) as { + allowed: boolean; + personal: { dailySpentUsd: number } | null; + provider: { provider: string; connectionId: string } | null; + }; + assert.equal(body.allowed, true); + assert.equal(body.personal?.dailySpentUsd, 1.25); + assert.equal(body.provider?.provider, "claude"); + assert.equal(body.provider?.connectionId, "conn-claude"); +}); + +test("om-usage without ?format stays text/plain (the historical contract)", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + assert.match(response.headers.get("content-type") ?? "", /text\/plain/); + const text = await response.text(); + assert.match(text, /Personal quota/); + assert.match(text, /Provider quota/); +}); + +test("om-usage ?format=json reports a disallowed key as structured allowed:false", async () => { + // A usage panel must tell "this key may not ask" apart from "no data yet", + // which a bare 403 text body cannot express. + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps({ + getApiKeyMetadata: async () => ({ id: "key-off", allowUsageCommand: false }), + }) + ); + + assert.equal(response.status, 403); + const body = (await response.json()) as { allowed: boolean }; + assert.equal(body.allowed, false); +}); + +test("om-usage ?format=json rejects an invalid key and never leaks a stack trace", async () => { + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-wrong" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 401); + const body = (await response.json()) as { allowed: boolean; error?: { message?: string } }; + assert.equal(body.allowed, false); + assert.ok( + !body.error?.message?.includes("at /"), + "error bodies must not carry stack frames (ERROR_SANITIZATION)" + ); +}); From 3ef54fc55b097c29caf319f2d4c28b10cede44de Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 22 Aug 2026 22:39:56 -0300 Subject: [PATCH 09/29] feat(api): return every connection's snapshot under providers[] in om-usage json (#11192) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(api): structured ?format=json for the self-service usage endpoint GET /api/usage/om-usage already let any key read its own usage — personal daily/weekly USD limits and the provider quota snapshot — but only as text/plain, which a UI cannot parse safely. OmniCopilot issue #8 asks exactly for this surface. Adds ?format=json, returning the ApiKeyUsageLimitStatus + UsageSnapshot the text is rendered from. Text and JSON share the same collectors (collectUsageSnapshots, getApiKeyUsageLimitStatus), so the two can never disagree about a number. The response is a discriminated union: a key without allowUsageCommand (403) or an invalid key (401) returns { allowed:false, error:{message} }, distinct from allowed:true with empty sections — the state a panel must render as "nothing learned yet", not a refusal. Text form unchanged; without ?format the contract is untouched. The endpoint was previously missing from API_REFERENCE.md; it now has a section documenting both forms, the allowUsageCommand gate, and the self-service auth model (caller's own key, not requireManagementAuth). Regression guards in tests/unit/usage-command-json-format.test.ts (4 tests: json shape, text default preserved, structured 403, sanitized 401 with no stack trace). Existing internal-usage-command suite still 12/12. * chore(changelog): correct the fragment to the real PR number (#11190) * feat(api): return every connection's snapshot under providers[] in om-usage json Closes #11191. buildUsageCommandJson picked a single snapshot via selectUsageSnapshot, so a panel could only ever show one provider. The collector already had them all — the single-pick is a presentation choice for a terminal. The JSON form now also returns the full UsageSnapshot[] alongside the selected provider, so a UI can render Codex / Claude / OpenCode side by side. The text form is untouched. --------- Co-authored-by: Xiangzhe --- .../11192-usage-command-providers-array.md | 1 + docs/reference/API_REFERENCE.md | 4 ++- src/lib/usage/internalUsageCommand.ts | 13 +++++--- tests/unit/usage-command-json-format.test.ts | 32 +++++++++++++++++++ 4 files changed, 44 insertions(+), 6 deletions(-) create mode 100644 changelog.d/features/11192-usage-command-providers-array.md diff --git a/changelog.d/features/11192-usage-command-providers-array.md b/changelog.d/features/11192-usage-command-providers-array.md new file mode 100644 index 00000000000..b7ef4211091 --- /dev/null +++ b/changelog.d/features/11192-usage-command-providers-array.md @@ -0,0 +1 @@ +- **feat(api):** `/api/usage/om-usage?format=json` now returns `providers[]` — every connection's quota snapshot, not just the single selected one — so a panel can render Codex / Claude / OpenCode side by side. The collector already gathered all of them; the single-pick `provider` field (kept) is a terminal presentation choice. Closes the per-connection gap from OmniCopilot #8 ([#11192](https://github.com/diegosouzapw/OmniRoute/pull/11192)) diff --git a/docs/reference/API_REFERENCE.md b/docs/reference/API_REFERENCE.md index 3e39f35ebaf..1b71551e373 100644 --- a/docs/reference/API_REFERENCE.md +++ b/docs/reference/API_REFERENCE.md @@ -663,7 +663,9 @@ refusal. On success: // present only when the key opted into per-key usage limits (daily/weekly USD): "personal": { "dailySpentUsd": 1.25, "dailyLimitUsd": 5, "dailyResetAtIso": "…", "weeklySpentUsd": 8, "weeklyLimitUsd": 20, "weeklyResetAtIso": "…" /* … */ }, // the selected provider quota snapshot, or null when nothing is cached yet: - "provider": { "connectionId": "…", "provider": "claude", "plan": "…", "quotas": { /* … */ } } + "provider": { "connectionId": "…", "provider": "claude", "plan": "…", "quotas": { /* … */ } }, + // every connection's snapshot, so a UI can render several providers side by side: + "providers": [ { "connectionId": "…", "provider": "claude", /* … */ }, { "provider": "codex", /* … */ } ] } ``` diff --git a/src/lib/usage/internalUsageCommand.ts b/src/lib/usage/internalUsageCommand.ts index 76d55315d9c..37f5f009d1c 100644 --- a/src/lib/usage/internalUsageCommand.ts +++ b/src/lib/usage/internalUsageCommand.ts @@ -546,6 +546,11 @@ export type UsageCommandJson = personal: unknown | null; /** The selected provider snapshot, or null when nothing is cached. */ provider: UsageSnapshot | null; + /** Every connection's snapshot, so a panel can render Codex / Claude / + * OpenCode side by side instead of only the selected one (#11191). The + * single-pick in `provider` is a presentation choice for a terminal; the + * collector already gathered all of them. */ + providers: UsageSnapshot[]; }; export async function buildUsageCommandJson( @@ -564,11 +569,9 @@ export async function buildUsageCommandJson( { now: resolvedDeps.now } ) : null; - const provider = selectUsageSnapshot( - await collectUsageSnapshots(metadata, resolvedDeps), - selection - ); - return { allowed: true, personal, provider }; + const snapshots = await collectUsageSnapshots(metadata, resolvedDeps); + const provider = selectUsageSnapshot(snapshots, selection); + return { allowed: true, personal, provider, providers: snapshots }; } export async function buildUsageCommandText( diff --git a/tests/unit/usage-command-json-format.test.ts b/tests/unit/usage-command-json-format.test.ts index f7d7b286c41..9f76d6e1913 100644 --- a/tests/unit/usage-command-json-format.test.ts +++ b/tests/unit/usage-command-json-format.test.ts @@ -40,6 +40,7 @@ function allowedDeps(overrides: Record = {}) { }), getProviderConnections: async () => [ { id: "conn-claude", provider: "claude", isActive: true }, + { id: "conn-codex", provider: "codex", isActive: true }, ], getAllProviderLimitsCache: () => ({ "conn-claude": { @@ -50,6 +51,14 @@ function allowedDeps(overrides: Record = {}) { message: null, fetchedAt: new Date(NOW).toISOString(), }, + "conn-codex": { + plan: "Codex Pro", + quotas: { + weekly: { used: 9, total: 100, remaining: 91, resetAt: "2026-08-24T03:00:00.000Z" }, + }, + message: null, + fetchedAt: new Date(NOW).toISOString(), + }, }), getProviderConnectionById: async () => null, getProviderLimitsCache: () => null, @@ -95,6 +104,29 @@ test("om-usage without ?format stays text/plain (the historical contract)", asyn assert.match(text, /Provider quota/); }); +test("om-usage ?format=json returns every connection under providers[], not just the selected one", async () => { + // #11191 — a panel needs Codex + Claude side by side; the single `provider` + // pick is a terminal presentation choice, the collector had them all. + const response = await handleInternalUsageCommandHttpRequest( + new Request("http://localhost/api/usage/om-usage?format=json", { + headers: { Authorization: "Bearer sk-allowed" }, + }), + allowedDeps() + ); + + assert.equal(response.status, 200); + const body = (await response.json()) as { + allowed: boolean; + provider: { provider: string } | null; + providers: Array<{ provider: string }>; + }; + assert.equal(body.allowed, true); + const names = body.providers.map((s) => s.provider).sort(); + assert.deepEqual(names, ["claude", "codex"]); + // the single-pick field is still present and one of them + assert.ok(["claude", "codex"].includes(body.provider?.provider ?? "")); +}); + test("om-usage ?format=json reports a disallowed key as structured allowed:false", async () => { // A usage panel must tell "this key may not ask" apart from "no data yet", // which a bare 403 text body cannot express. From eb5797370ae92097d277a783f2d517ccb026dd91 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 22 Aug 2026 22:46:34 -0300 Subject: [PATCH 10/29] feat(dashboard): add CheaperInference sponsor banner and route banner links through the shortener (#11196) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - New CheaperInferenceSponsorBanner on the dashboard home, same size/shape as KimiSponsorBanner, no version gate (durable partnership). Uses the cheaperinference ProviderIcon and the brand green (#31f889) with the dark ink CTA (contrast, per colors.ts token). - CTA points at https://link.omniroute.online/cheaper — the branded short link — so clicks land in our Kutt metrics. - VscodeCopilotBanner CTA now points at https://link.omniroute.online/vsx instead of the raw Marketplace URL, for the same reason. - i18n strings in en + pt (en is the namespace-level fallback for the other 41 locales). - Tests: new cheaperInferenceSponsorBanner.test.tsx (render, CTA href, dismiss persistence); vscodeCopilotBanner.test.tsx updated to the new CTA URL. Co-authored-by: Xiangzhe --- .../CheaperInferenceSponsorBanner.tsx | 108 ++++++++++++++++++ .../dashboard/VscodeCopilotBanner.tsx | 4 +- src/app/(dashboard)/home/page.tsx | 2 + src/i18n/messages/en.json | 7 ++ src/i18n/messages/pt.json | 7 ++ .../ui/cheaperInferenceSponsorBanner.test.tsx | 92 +++++++++++++++ tests/unit/ui/vscodeCopilotBanner.test.tsx | 2 +- 7 files changed, 220 insertions(+), 2 deletions(-) create mode 100644 src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx create mode 100644 tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx diff --git a/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx new file mode 100644 index 00000000000..77cf1932983 --- /dev/null +++ b/src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner.tsx @@ -0,0 +1,108 @@ +"use client"; + +import { useSyncExternalStore } from "react"; +import { useTranslations } from "next-intl"; +import ProviderIcon from "@/shared/components/ProviderIcon"; + +// Branded short link through our own link.omniroute.online shortener, so the +// click lands in our Kutt metrics. Points at cheaperinference.com?utm_source=omniroute +// (the URL in README.md's Open Source Friends section). Keep in sync with the +// `cheaper` slug on the shortener. +const CHEAPER_INFERENCE_URL = "https://link.omniroute.online/cheaper"; + +// Cheaper Inference brand green (#31f889). White text on it fails contrast, so +// the CTA pairs it with the dark ink from the provider's color token (colors.ts: +// cheaperinference.text = #04170d). Hex values stay in sync with that token. + +const DISMISS_STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +// Same-tab signal for the dismiss button, since writing localStorage doesn't +// fire a "storage" event in the tab that wrote it. +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; + +function isNotDismissed(): boolean { + try { + return !localStorage.getItem(DISMISS_STORAGE_KEY); + } catch { + return true; + } +} + +function subscribe(callback: () => void) { + window.addEventListener(DISMISS_EVENT, callback); + return () => window.removeEventListener(DISMISS_EVENT, callback); +} + +// SSR has no localStorage, so the server always renders the banner visible; +// useSyncExternalStore reconciles that against the real client-side value +// right after hydration, mirroring KimiSponsorBanner's pattern. +function getServerSnapshot() { + return true; +} + +/** + * Dismissable banner announcing the Cheaper Inference OmniRoute partnership on + * the dashboard home page — same size/shape as KimiSponsorBanner, no version + * gate (durable partnership, not a time-boxed offer). The logomark reuses + * . + */ +export default function CheaperInferenceSponsorBanner() { + const t = useTranslations("cheaperInferenceSponsorBanner"); + const visible = useSyncExternalStore(subscribe, isNotDismissed, getServerSnapshot); + + if (!visible) { + return null; + } + + const dismiss = () => { + try { + localStorage.setItem(DISMISS_STORAGE_KEY, "true"); + } catch { + // ignore — worst case the banner reappears next visit + } + window.dispatchEvent(new Event(DISMISS_EVENT)); + }; + + return ( +
+
+
+ +
+
+

{t("title")}

+

{t("description")}

+
+
+ +
+
+ + {t("cta")} + + + {t("partnerLinkNote")} +
+ +
+
+ ); +} diff --git a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx index 0b23976531c..1715f17c839 100644 --- a/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx +++ b/src/app/(dashboard)/dashboard/VscodeCopilotBanner.tsx @@ -6,7 +6,9 @@ import { useTranslations } from "next-intl"; // Marketplace listing is the primary CTA; Open VSX (Cursor/Windsurf/VSCodium/etc.) // is called out via secondaryNote instead of a second button, to keep this banner // the same size as KimiSponsorBanner. -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +// Branded short link through our own link.omniroute.online shortener (the `vsx` +// slug), so the click lands in our Kutt metrics. +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; const DISMISS_STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; // Same-tab signal for the dismiss button, since writing localStorage doesn't diff --git a/src/app/(dashboard)/home/page.tsx b/src/app/(dashboard)/home/page.tsx index beccde22610..bc10df88f42 100644 --- a/src/app/(dashboard)/home/page.tsx +++ b/src/app/(dashboard)/home/page.tsx @@ -4,6 +4,7 @@ import { getSettings } from "@/lib/localDb"; import HomePageClient from "../dashboard/HomePageClient"; import BootstrapBanner from "../dashboard/BootstrapBanner"; import KimiSponsorBanner from "../dashboard/KimiSponsorBanner"; +import CheaperInferenceSponsorBanner from "../dashboard/CheaperInferenceSponsorBanner"; import VscodeCopilotBanner from "../dashboard/VscodeCopilotBanner"; import NewsBanner from "../dashboard/NewsBanner"; @@ -20,6 +21,7 @@ export default async function HomePage() { <> {isBootstrapped && } + diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index ff778753c61..930f5b59587 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -13869,5 +13869,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "Cheaper Inference is an OmniRoute Open Source Friend", + "description": "A cost-ranked gateway reselling dozens of frontier models behind one OpenAI-compatible endpoint — routing each request to the cheapest eligible provider, never above list price.", + "cta": "Get an API Key", + "partnerLinkNote": "Partner link", + "dismissAriaLabel": "Dismiss" } } diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 11a3606a54c..1921f3fba7f 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -13845,5 +13845,12 @@ "toolsMismatch": "Provider does not support tool calling", "structuredOutputMismatch": "Provider does not support structured output", "contextWindowMismatch": "Request exceeds provider context window" + }, + "cheaperInferenceSponsorBanner": { + "title": "A Cheaper Inference é uma Amiga do Código Aberto do OmniRoute", + "description": "Um gateway com custo ordenado que revende dezenas de modelos de fronteira num único endpoint compatível com OpenAI — roteando cada requisição ao provedor elegível mais barato, nunca acima do preço de tabela.", + "cta": "Obter uma Chave de API", + "partnerLinkNote": "Link de parceiro", + "dismissAriaLabel": "Dispensar" } } diff --git a/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx b/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx new file mode 100644 index 00000000000..b86f4188c6a --- /dev/null +++ b/tests/unit/ui/cheaperInferenceSponsorBanner.test.tsx @@ -0,0 +1,92 @@ +// @vitest-environment jsdom +/** + * CheaperInferenceSponsorBanner — render gate (localStorage dismissal), CTA + * pointing at our link.omniroute.online branded short link, and discreet + * partner-link note. Mirrors kimiSponsorBanner.test.tsx, minus the version gate + * (this banner is a durable partnership, not a time-boxed offer). + */ +import React from "react"; +import { act } from "react"; +import { createRoot } from "react-dom/client"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; + +const STORAGE_KEY = "omniroute-cheaperinference-sponsor-banner-dismissed-v1"; +const DISMISS_EVENT = "omniroute:cheaperinference-sponsor-banner-dismissed"; +const SHORT_URL = "https://link.omniroute.online/cheaper"; + +vi.mock("next-intl", () => ({ useTranslations: () => (k: string) => k })); +vi.mock("@/shared/components/ProviderIcon", () => ({ default: () => null })); + +async function renderBanner(): Promise { + vi.resetModules(); + const { default: CheaperInferenceSponsorBanner } = + await import("../../../src/app/(dashboard)/dashboard/CheaperInferenceSponsorBanner"); + + const container = document.createElement("div"); + document.body.appendChild(container); + const root = createRoot(container); + act(() => { + root.render(); + }); + return container; +} + +describe("CheaperInferenceSponsorBanner", () => { + beforeEach(() => { + ( + globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean } + ).IS_REACT_ACT_ENVIRONMENT = true; + localStorage.removeItem(STORAGE_KEY); + }); + + afterEach(() => { + document.body.innerHTML = ""; + localStorage.removeItem(STORAGE_KEY); + }); + + it("renders with the CTA pointing at the branded short link", async () => { + const container = await renderBanner(); + expect(container.textContent).toContain("title"); + expect(container.textContent).toContain("cta"); + const link = container.querySelector("a[href]"); + expect(link).not.toBeNull(); + expect(link?.getAttribute("href")).toBe(SHORT_URL); + expect(link?.getAttribute("target")).toBe("_blank"); + expect(link?.getAttribute("rel")).toContain("noopener"); + }); + + it("shows the discreet partner-link note near the CTA", async () => { + const container = await renderBanner(); + expect(container.textContent).toContain("partnerLinkNote"); + const link = container.querySelector("a[href]"); + expect(link?.getAttribute("title")).toBe("partnerLinkNote"); + }); + + it("hides after dismissal and stays hidden on re-render", async () => { + const first = await renderBanner(); + const button = first.querySelector("button"); + expect(button).not.toBeNull(); + act(() => { + button?.click(); + }); + expect(localStorage.getItem(STORAGE_KEY)).toBe("true"); + expect(first.textContent).not.toContain("title"); + + // a fresh render (simulating a later visit) stays hidden + const second = await renderBanner(); + expect(second.textContent).not.toContain("title"); + }); + + it("re-renders visible again only after the key is cleared", async () => { + const first = await renderBanner(); + const button = first.querySelector("button"); + act(() => { + button?.click(); + }); + expect(localStorage.getItem(STORAGE_KEY)).toBe("true"); + + localStorage.removeItem(STORAGE_KEY); + const second = await renderBanner(); + expect(second.textContent).toContain("title"); + }); +}); diff --git a/tests/unit/ui/vscodeCopilotBanner.test.tsx b/tests/unit/ui/vscodeCopilotBanner.test.tsx index ed75ad4950a..fad50b2b266 100644 --- a/tests/unit/ui/vscodeCopilotBanner.test.tsx +++ b/tests/unit/ui/vscodeCopilotBanner.test.tsx @@ -11,7 +11,7 @@ import { createRoot } from "react-dom/client"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; const STORAGE_KEY = "omniroute-vscode-copilot-banner-dismissed-v1"; -const MARKETPLACE_URL = "https://marketplace.visualstudio.com/items?itemName=diegosouzapw.omnicopilot"; +const MARKETPLACE_URL = "https://link.omniroute.online/vsx"; vi.mock("next-intl", () => ({ useTranslations: () => (k: string) => k })); From d888f1a08b109ade858d085b5403a7116209f1c1 Mon Sep 17 00:00:00 2001 From: amrx Date: Sat, 22 Aug 2026 18:49:08 -0700 Subject: [PATCH 11/29] fix(combo): accept SSE comment lines (e.g. OpenRouter keep-alives) in response quality validation (#11036) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SSE comment lines (OpenRouter keep-alives) and leading whitespace no longer fail the combo quality gate's JSON fallback. First contribution — clean minimal fix with test. Thank you @asorourx, welcome aboard! --- open-sse/services/combo/validateQuality.ts | 13 ++++++++++++- tests/unit/validate-response-quality.test.ts | 14 ++++++++++++++ 2 files changed, 26 insertions(+), 1 deletion(-) diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index 7c7f0ce8a92..76ef5546aba 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -617,7 +617,18 @@ export async function validateResponseQuality( try { json = JSON.parse(text); } catch { - if (text.startsWith("data:") || text.startsWith("event:")) return { valid: true }; + // An SSE stream body is expected for streamed upstreams. Besides `data:` and + // `event:` frames, the SSE spec also allows comment lines that begin with a + // colon (`:`), which providers use for keep-alives while the model is still + // generating — e.g. OpenRouter emits `: OPENROUTER PROCESSING` on slower / + // reasoning responses. A stream that opens with such a comment (or with + // leading whitespace/newlines) is still a valid stream, not malformed JSON, + // so trim and recognize the comment prefix before rejecting. Without this, + // otherwise-good streamed completions get failed as "not valid JSON". + const trimmed = text.trimStart(); + if (trimmed.startsWith("data:") || trimmed.startsWith("event:") || trimmed.startsWith(":")) { + return { valid: true }; + } return { valid: false, reason: "response is not valid JSON" }; } diff --git a/tests/unit/validate-response-quality.test.ts b/tests/unit/validate-response-quality.test.ts index de4b5f47dc3..6a918155e56 100644 --- a/tests/unit/validate-response-quality.test.ts +++ b/tests/unit/validate-response-quality.test.ts @@ -24,6 +24,20 @@ test("returns valid=true for SSE with 'data:' lines", async () => { assert.strictEqual(res.valid, true); }); +test("returns valid=true for SSE opening with a ':' comment line (e.g. OpenRouter keep-alive)", async () => { + const res = await validateResponseQuality( + makeResponse(': OPENROUTER PROCESSING\n\ndata: {"foo":"bar"}\n\n'), + false, + {} + ); + assert.strictEqual(res.valid, true); +}); + +test("returns valid=true for an SSE stream with leading whitespace before the first frame", async () => { + const res = await validateResponseQuality(makeResponse('\n\ndata: {"foo":"bar"}\n\n'), false, {}); + assert.strictEqual(res.valid, true); +}); + test("returns valid=false for non-JSON non-SSE text", async () => { const res = await validateResponseQuality(makeResponse("Hello world"), false, {}); assert.strictEqual(res.valid, false); From e73ab0040cf33d6fafd45562b45558cca42f03ad Mon Sep 17 00:00:00 2001 From: Ke Jin Date: Sun, 23 Aug 2026 09:49:11 +0800 Subject: [PATCH 12/29] fix(codex): make remote compaction V2 complete reliably (#11041) Compaction-V2 output now counts as real model output (no synthetic response.failed after response.completed), the Codex SSE filter handles CRLF framing, and terminal detection runs before scan-state bounding. 88/88 stream/readiness suites on the board. Thank you @jackjinke! --- open-sse/executors/codex.ts | 10 +-- open-sse/utils/streamHandler.ts | 8 ++- open-sse/utils/streamReadiness.ts | 8 +++ .../codex-drop-nonstandard-events.test.ts | 40 ++++++++++-- .../unit/empty-stream-no-content-8649.test.ts | 65 +++++++++++++++++++ tests/unit/stream-handler.test.ts | 35 ++++++++++ 6 files changed, 152 insertions(+), 14 deletions(-) diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index e1674c30fb3..6e39bb81caa 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -517,10 +517,12 @@ export function filterNonstandardCodexSse(response: Response): Response { const transform = new TransformStream({ transform(chunk, controller) { buffer += decoder.decode(chunk, { stream: true }); - let sep: number; - while ((sep = buffer.indexOf("\n\n")) !== -1) { - const block = buffer.slice(0, sep + 2); - buffer = buffer.slice(sep + 2); + while (true) { + const separator = /\r?\n\r?\n/.exec(buffer); + if (!separator) break; + const blockEnd = separator.index + separator[0].length; + const block = buffer.slice(0, blockEnd); + buffer = buffer.slice(blockEnd); if (!dropBlock(block)) controller.enqueue(encoder.encode(block)); } }, diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index 4d8993c7670..9adb5dbf2f6 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -683,13 +683,15 @@ export function createDisconnectAwareStream(transformStream, streamController) { if (clientTerminalSeen) return; terminalTail += terminalDecoder.decode(chunk, { stream: true }); - if (terminalTail.length > 4096) { - terminalTail = terminalTail.slice(-4096); - } + // Scan before bounding retained state: a compaction terminal frame can + // exceed the tail budget because encrypted_content is carried inline. clientTerminalSeen = hasClientTerminalSseMarker( terminalTail, streamController.clientResponseFormat ); + if (terminalTail.length > 4096) { + terminalTail = terminalTail.slice(-4096); + } if (clientTerminalSeen) { streamController.markClientTerminalSeen?.(); } diff --git a/open-sse/utils/streamReadiness.ts b/open-sse/utils/streamReadiness.ts index 2d06659b809..1696a7a5c52 100644 --- a/open-sse/utils/streamReadiness.ts +++ b/open-sse/utils/streamReadiness.ts @@ -34,6 +34,14 @@ function hasUsefulValue(value: unknown): boolean { if (Array.isArray(value)) return value.some(hasUsefulValue); if (!isRecord(value)) return false; + // A Responses compaction item IS the turn's output: remote compaction + // completes with output = [{type:"compaction", encrypted_content}] and no + // assistant text. Deliberately NOT a blanket encrypted_content key — an + // encrypted reasoning item alone is not user-visible output and must keep + // tripping the #8649 empty-content guard. + // This shape is specific to Responses streams; chat-completion frames do not produce it. + if (value.type === "compaction" && hasNonEmptyString(value.encrypted_content)) return true; + for (const key of [ "content", "text", diff --git a/tests/unit/codex-drop-nonstandard-events.test.ts b/tests/unit/codex-drop-nonstandard-events.test.ts index d2aff548149..dec24970ca5 100644 --- a/tests/unit/codex-drop-nonstandard-events.test.ts +++ b/tests/unit/codex-drop-nonstandard-events.test.ts @@ -17,6 +17,19 @@ function sseResponse(body: string): Response { }); } +function chunkedSseResponse(chunks: string[]): Response { + const encoder = new TextEncoder(); + return new Response( + new ReadableStream({ + start(controller) { + for (const chunk of chunks) controller.enqueue(encoder.encode(chunk)); + controller.close(); + }, + }), + { status: 200, headers: { "content-type": "text/event-stream" } } + ); +} + async function readAll(res: Response): Promise { return await res.text(); } @@ -61,10 +74,10 @@ describe("codexDropNonstandardEvents (#11014)", () => { describe("filterNonstandardCodexSse (#4715)", () => { it("drops codex.* event blocks but keeps standard response.* events", async () => { const stream = - "event: response.created\ndata: {\"type\":\"response.created\"}\n\n" + + 'event: response.created\ndata: {"type":"response.created"}\n\n' + "event: codex.rate_limits\n\n" + - "event: response.output_text.delta\ndata: {\"delta\":\"hi\"}\n\n" + - "event: response.completed\ndata: {\"type\":\"response.completed\"}\n\n"; + 'event: response.output_text.delta\ndata: {"delta":"hi"}\n\n' + + 'event: response.completed\ndata: {"type":"response.completed"}\n\n'; const out = await readAll(filterNonstandardCodexSse(sseResponse(stream))); assert.ok(!out.includes("codex.rate_limits"), "codex.* frame must be stripped"); assert.ok(out.includes("response.created"), "standard events preserved"); @@ -72,18 +85,31 @@ describe("filterNonstandardCodexSse (#4715)", () => { assert.ok(out.includes("response.completed"), "terminal event preserved"); }); + it("filters CRLF-framed events split across transport chunks", async () => { + const response = chunkedSseResponse([ + 'event: response.created\r\ndata: {"type":"response.created"}\r\n\r', + "\nevent: codex.rate_limits\r\n\r\n", + 'event: response.completed\r\ndata: {"type":"response.completed"}\r\n\r\n', + ]); + + const out = await readAll(filterNonstandardCodexSse(response)); + + assert.ok(!out.includes("codex.rate_limits"), "codex.* frame must be stripped"); + assert.ok(out.includes("response.created"), "standard events preserved"); + assert.ok(out.includes("response.completed"), "terminal event preserved"); + }); + it("passes through non-SSE responses untouched", async () => { - const json = new Response("{\"ok\":true}", { + const json = new Response('{"ok":true}', { status: 200, headers: { "content-type": "application/json" }, }); const out = filterNonstandardCodexSse(json); - assert.equal(await out.text(), "{\"ok\":true}"); + assert.equal(await out.text(), '{"ok":true}'); }); it("drops a trailing codex.* block with no double-newline terminator (flush path)", async () => { - const stream = - "event: response.created\ndata: {}\n\n" + "event: codex.token_count\ndata: {}"; + const stream = "event: response.created\ndata: {}\n\n" + "event: codex.token_count\ndata: {}"; const out = await readAll(filterNonstandardCodexSse(sseResponse(stream))); assert.ok(out.includes("response.created")); assert.ok(!out.includes("codex.token_count")); diff --git a/tests/unit/empty-stream-no-content-8649.test.ts b/tests/unit/empty-stream-no-content-8649.test.ts index e46157fbe17..918ae7c4c06 100644 --- a/tests/unit/empty-stream-no-content-8649.test.ts +++ b/tests/unit/empty-stream-no-content-8649.test.ts @@ -252,3 +252,68 @@ test("#8649 buildStreamErrorChunks-shaped error must not be rewritten as empty c assert.match(text, /AI Model Not Found/); assert.doesNotMatch(text, /Provider returned empty content/); }); + +test("#8649 a Responses compaction-only stream is real output, not empty content", async () => { + // Codex remote compaction V2: POST /v1/responses with a compaction_trigger + // input item completes with output = [{type:"compaction", encrypted_content}] + // and no assistant text. The watcher's content keys do not include + // encrypted_content, so the healthy stream was followed by a synthetic + // response.failed ("Provider returned empty content") — which strict + // Responses clients reject even after response.completed. + const text = await runClientStream( + [ + `data: {"type":"response.in_progress"}\n\n`, + `event: response.created\ndata: ${JSON.stringify({ + type: "response.created", + response: { id: "resp_cmp", status: "in_progress", output: [] }, + })}\n\n`, + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp_cmp", + status: "completed", + output: [ + { id: "cmp_1", type: "compaction", encrypted_content: "gAAAAABencryptedpayload" }, + ], + }, + })}\n\n`, + ], + FORMATS.OPENAI_RESPONSES + ); + + assert.match(text, /"type":"compaction"/); + assert.doesNotMatch( + text, + /Provider returned empty content|response\.failed/, + "a completed compaction response must not be followed by a synthetic failure frame" + ); +}); + +test("#8649 an encrypted-reasoning-only stream is still empty content", async () => { + // Inverse of the compaction carve-out: an encrypted reasoning item is not + // user-visible output. A turn that produces only a reasoning trace and no + // message/tool call is the fake-success shape this guard exists to catch. + const text = await runClientStream( + [ + `event: response.created\ndata: ${JSON.stringify({ + type: "response.created", + response: { id: "resp_r", status: "in_progress", output: [] }, + })}\n\n`, + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp_r", + status: "completed", + output: [{ id: "rs_1", type: "reasoning", encrypted_content: "gAAAAABencryptedtrace" }], + }, + })}\n\n`, + ], + FORMATS.OPENAI_RESPONSES + ); + + assert.match( + text, + /response\.failed|Provider returned empty content/, + "a reasoning-only turn must keep tripping the empty-content guard" + ); +}); diff --git a/tests/unit/stream-handler.test.ts b/tests/unit/stream-handler.test.ts index d19a13f351e..8c19c088021 100644 --- a/tests/unit/stream-handler.test.ts +++ b/tests/unit/stream-handler.test.ts @@ -167,6 +167,41 @@ test("createDisconnectAwareStream treats cancel after Responses completed as suc assert.equal(disconnectHandled, false); }); +test("createDisconnectAwareStream recognizes a large Responses compaction completion", async () => { + let errorHandled = false; + const completed = `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + status: "completed", + output: [{ type: "compaction", encrypted_content: "x".repeat(5000) }], + }, + })}\n\n`; + const transformStream = { + readable: new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(completed)); + controller.close(); + }, + }), + writable: createNoopAbortWritable(), + }; + + const stream = createDisconnectAwareStream( + transformStream, + createStreamController({ + clientResponseFormat: FORMATS.OPENAI_RESPONSES, + onError() { + errorHandled = true; + }, + }) + ); + const text = await readStreamText(stream); + + assert.equal(text, completed); + assert.equal(errorHandled, false); + assert.doesNotMatch(text, /response\.failed/); +}); + test("createDisconnectAwareStream: Gemini 503 high-demand error becomes SSE error chunk with message preserved", async () => { const geminiMsg = "[503]: This model is currently experiencing high demand. Spikes in demand are usually temporary. Please try again later."; From 230017196cd158b5a69781d0ededa567dc3fe302 Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Sun, 23 Aug 2026 03:51:02 +0200 Subject: [PATCH 13/29] fix(resilience): drain heavyweight SSE on SIGTERM (#11020) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board + this branch alone: chat-body-admission + authz/pipeline 65/65, file-size gate green with a dated frozen entry (chatBodyAdmission 1005→1009 — the +4 lease/drain wiring lines, owner-authorized rebaseline). trackRequest was never called, so SIGTERM waitForDrain saw zero in-flight and killed live SSE; leases now hold the drain counter for the stream's lifetime, and the 503 carries Retry-After. Closes #11015. Thank you @RaviTharuma! --- changelog.d/fixes/11015-shutdown-track-sse.md | 1 + config/quality/file-size-baseline.json | 8 +++-- docs/guides/DOCKER_GUIDE.md | 31 ++++++++++++++++++- src/server/authz/pipeline.ts | 1 + src/shared/middleware/chatBodyAdmission.ts | 5 +++ tests/unit/authz/pipeline.test.ts | 2 ++ tests/unit/chat-body-admission.test.ts | 20 ++++++++++++ 7 files changed, 65 insertions(+), 3 deletions(-) create mode 100644 changelog.d/fixes/11015-shutdown-track-sse.md diff --git a/changelog.d/fixes/11015-shutdown-track-sse.md b/changelog.d/fixes/11015-shutdown-track-sse.md new file mode 100644 index 00000000000..1ed99b3669d --- /dev/null +++ b/changelog.d/fixes/11015-shutdown-track-sse.md @@ -0,0 +1 @@ +- **fix(resilience):** count heavyweight `/v1` admission leases in the SIGTERM drain and send `Retry-After` on shutdown 503s so Recreate no longer looks like an empty 502 ([#11015](https://github.com/diegosouzapw/OmniRoute/issues/11015)) — thanks @RaviTharuma diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index f803789ffde..2eb468e16c6 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,5 +1,6 @@ { "_rebaseline_2026_08_20_10531_freebuff_provider": "PR #10531 (adrianaryaputra, feat/freebuff-provider-support, closes #6793) own growth: src/shared/constants/providers/apikey/gateways.ts 1283->1298 (+15, the freebuff APIKEY_PROVIDERS_GATEWAYS catalog entry, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines) and src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx 1062->1067 (+5, freebuff credential placeholder/hint at the existing per-provider switch chokepoint). Covered by tests/unit/freebuff-provider.test.ts (9/9 passing).", + "_rebaseline_2026_08_21_10987_logfare_provider": "PR #10987 (jonlwheat2-gif, feat/10644-logfare-provider, closes #10644) own growth: src/shared/constants/providers/apikey/gateways.ts 1298->1321 (+23, the logfare APIKEY_PROVIDERS_GATEWAYS catalog entry with Free badge/freeNote/apiHint documenting the request-logging policy, additive data at the existing registry chokepoint, same god-file no-split rationale as the prior gateways.ts rebaselines: #10531 freebuff, merge-storm 2026-08-11). Covered by tests/unit/logfare-registry.test.ts (1/1 passing).", "_rebaseline_2026_08_20_10574_reasoning_transport_fallback": "PR #10574 (jackjinke, fix/responses-reasoning-transport, fixes #10550) own growth: src/sse/handlers/chatHelpers.ts 1017->1019 (+2 = the new reasoningTransportFallback option threaded through executeChatWithBreaker's options destructure and its downstream handleSingleModel call, at the existing per-attempt options-passthrough chokepoint; not extractable without splitting the option-forwarding call itself). Covered by the PR's own reasoning-policy test suite (tests/unit/chatcore-translation-paths.test.ts, tests/unit/combo-attempt-body-isolation-7847.test.ts, tests/unit/reasoning-cache.test.ts, tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts among others), 446/446 focused tests passing.", "_rebaseline_2026_08_18_10517_zed_hosted_oauth_callback_port": "PR #10517 (phatchau036, fix/zed-hosted-oauth-callback-port) own growth: src/shared/components/OAuthModal.tsx 1131->1148 (wc -l; check-file-size.mjs counts via split(\"\\n\").length so the gate sees 1134->1149, +15/+18, crosses the frozen 1134 cap). Wires the zed-hosted native-app callback auto-complete: forceManual gating on isTrueLocalhost for zed-hosted, the loopback-redirect-URI comment block, and the exchangeToken full-URL-as-code branch, all at the existing provider-switch chokepoints this modal already carries growth for (seventh bump: 969->989->993->998->1030->1056->1100->1149; structural shrink tracked in #3501). The actual port-derivation logic lives in src/lib/oauth/providers/zed-hosted.ts (not frozen here) and was hardened during pre-merge review to use the server's own getRuntimePorts() instead of a browser-guessed scheme/port, covered by the new tests/unit/zed-hosted-loopback-port-derivation.test.ts (8/8 passing).", "_rebaseline_2026_08_13_10243_codex_fingerprint_merge": "PR #10243 (xz-dev, Codex OAuth fingerprint convergence) merge into release/v3.8.50: src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts crossed the 1000-line new-file cap for the first time (974 on base, 997 on the PR's own branch, 1013 after merging + prettier reflow) purely from combining two independent, already-legitimate feature additions that landed on the same shared UI-helper file — this PR's own Codex fingerprint-mode select/toggle wiring (CODEX_FINGERPRINT_MODE_VALUES, getCodexFingerprintModeLabel, CodexFingerprintModeValue) plus #8949's unrelated Codex account-service-tier helpers merged concurrently on release/v3.8.50. Neither addition alone crosses the cap; git's line-level auto-merge does not detect a threshold crossing. Not modularized as part of this conflict-resolution merge commit (out of scope — this is a merge, not a feature change). Covered by the PR's own tests/unit/codex-fingerprint-convergence.test.ts, tests/unit/executor-codex.test.ts, tests/unit/provider-specific-data-schema.test.ts (all passing post-merge).", @@ -388,6 +389,8 @@ "open-sse/services/claudeCodeCompatible.ts": 1563, "open-sse/services/combo.ts": 4742, "open-sse/services/compression/strategySelector.ts": 1379, + "open-sse/services/compression/engines/ccr/index.ts": 1024, + "_rebaseline_2026_08_22_11084_ccr_caller_gate": "PR #11084 (HouMinXi) own growth: open-sse/services/compression/engines/ccr/index.ts 1000->1024 (first listing — the engine was unlisted and drifted just over the 1000 cap; +24 are the callerSupportsCcrRetrieve gate that skips replacement entirely for callers without the retrieve tool, closing the stranded-prompt incident measured in production). Covered by tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "open-sse/services/contextManager.ts": 1001, "_rebaseline_2026_08_22_11113_purify_system_first": "PR #11113 (ggdayup) own growth: open-sse/services/contextManager.ts 1000->1001 (+1, purifyHistory merges the compression notice into the leading system message instead of splicing a second one mid-array — live-confirmed TokenRouter 400s; the +1 is the merge-into-leading branch, not extractable). Covered by tests/unit/context-manager-purify-system-first.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "open-sse/services/rateLimitManager.ts": 1517, @@ -449,7 +452,7 @@ "_rebaseline_2026_08_22_11156_enter_check_disabled": "PR #11156 (rqzbeh) own growth: AddApiKeyModal.tsx 1080->1082 (+2, Enter keydown handler now mirrors the isCheckDisabled condition — owner-requested post-merge polish from #11056; the rest of the diff is Prettier reflow). Covered by tests/unit/ui/add-api-key-modal-enter-key.test.tsx (jsdom render test, Enter dispatch assertions).", "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051, "src/shared/components/ModelSelectModal.tsx": 1138, - "src/shared/constants/providers/apikey/gateways.ts": 1298, + "src/shared/constants/providers/apikey/gateways.ts": 1321, "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1387, "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry": "DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legitima acima do cap; gateways.ts = god-file de catalogo de providers que cresceu com PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o proprio PR #9421 quebrou o arquivo); bridge.ts (PR #8949) = ponte Chromium vendored; proxyFetch.ts 1207->1220 = drift herdado de merges. Owner autorizou rebaseline com anotacao (2026-08-11).", "src/lib/modelCapabilities.ts": 1072, @@ -457,7 +460,8 @@ "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 1014, "open-sse/config/imageRegistry.ts": 1034, "src/sse/handlers/chatHelpers.ts": 1019, - "src/shared/middleware/chatBodyAdmission.ts": 1005, + "src/shared/middleware/chatBodyAdmission.ts": 1009, + "_rebaseline_2026_08_22_11020_sigterm_drain": "PR #11020 (RaviTharuma) own growth: chatBodyAdmission.ts 1005->1009 (+4, heavyweight admission leases now increment the SIGTERM drain counter and releaseChatAdmissionWhenDone holds it for the SSE lifetime — closes #11015; +4 are the lease/drain wiring lines at the existing admission chokepoint). Covered by tests/unit/chat-body-admission.test.ts heavyweight-lease cases. Owner pre-authorized baseline bumps 2026-08-22.", "_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file).", "open-sse/executors/commandCode.ts": 1059, "_rebaseline_2026_08_21_10859_vision_bridge_catalog": "#10859 own growth (Vision Bridge fixes #10808/#10809): src/lib/modelCapabilities.ts 1006->1016 (+10, cmd/gpt-5.3-codex* text-only capability resolution) and open-sse/executors/commandCode.ts 988->1023 (+35, Command Code wire-model normalization for bare ids + reasoning field fallback for opencode-routed gateways). Cohesive bug fixes at the existing capability-resolution / executor chokepoints; not extractable mid-fix. Covered by tests/unit/model-capabilities-command-code-codex-textonly-10703.test.ts, tests/unit/command-code-vision.test.ts, tests/unit/opencode-mimo-reasoning-details-nonstream.test.ts. Pushed directly to release (own-session miss: the original rebaseline was made in a throwaway validation worktree and never landed on the PR branch or the release before merge).", diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 61d414de272..b24e7b7f610 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -505,7 +505,7 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. | Constraint | Consequence | | --- | --- | | Single writer | Do **not** run multiple replicas against the same SQLite file. That corrupts the DB. | -| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. | +| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. New requests during the empty-endpoint window get a reverse-proxy **`502 Bad Gateway: Unknown error`**, not OmniRoute JSON — clients cannot distinguish this from a provider failure (#11015). | | Same event loop as `/healthz` | A busy catalog or compression tick can delay probes; a short timeout then restarts the **only** replica. | **Probe matrix** (see also [Kubernetes probe recommendations](../ops/MONITORING_GUIDE.md#kubernetes-probe-recommendations)): @@ -518,6 +518,35 @@ Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. **Upgrades:** expect every session to drop. Drain clients if you can; there is no rolling update on default SQLite. Compose `restart: unless-stopped` plus Docker `HEALTHCHECK` will also replace the only process when the container is Unhealthy — same blast radius. +Kubernetes snippet for a **single replica** (Recreate is required; do not raise `replicas` against one SQLite file): + +```yaml +spec: + replicas: 1 + strategy: + type: Recreate + template: + spec: + terminationGracePeriodSeconds: 90 + containers: + - name: omniroute + lifecycle: + preStop: + exec: + command: ["/bin/sleep", "15"] + readinessProbe: + httpGet: + path: /healthz + port: 20128 + periodSeconds: 5 + livenessProbe: + tcpSocket: + port: 20128 + periodSeconds: 20 +``` + +`preStop` sleep lets kube drop Service endpoints before SIGTERM so **new** traffic stops hitting the dying process. In-flight `/v1/responses` SSE is drained up to `SHUTDOWN_TIMEOUT_MS` (default 30s) via heavyweight admission leases (#11015). New requests that still reach the process get `503` + `Retry-After: 5`. The Recreate empty-endpoint gap until the replacement is Ready remains a hard outage — that is the SQLite topology, not a probe misconfig. + External Postgres / multi-writer HA is **not** a documented stock path. If you need HA, keep a single replica or run a topology the project has tested and documented separately. The Postgres/MySQL work lives in [#8075](https://github.com/diegosouzapw/OmniRoute/issues/8075). Until that ships, the only supported way to multiply **large** `/v1/responses` capacity is N independent processes (next section), not `replicas > 1` on one volume. ## Scale-out: N independent processes diff --git a/src/server/authz/pipeline.ts b/src/server/authz/pipeline.ts index 9f4e46bed1e..d8dacf376d7 100644 --- a/src/server/authz/pipeline.ts +++ b/src/server/authz/pipeline.ts @@ -205,6 +205,7 @@ function drainingResponse(requestId: string): NextResponse { { status: 503 } ); response.headers.set(AUTHZ_HEADER_REQUEST_ID, requestId); + response.headers.set("Retry-After", "5"); return response; } diff --git a/src/shared/middleware/chatBodyAdmission.ts b/src/shared/middleware/chatBodyAdmission.ts index 1c3c25b904b..9b20d78e5ae 100644 --- a/src/shared/middleware/chatBodyAdmission.ts +++ b/src/shared/middleware/chatBodyAdmission.ts @@ -18,6 +18,7 @@ import { CORS_HEADERS } from "../utils/cors"; import { createHmac } from "crypto"; import v8 from "node:v8"; +import { trackRequest } from "../../lib/gracefulShutdown"; function parsePositiveInt(value: string | undefined, fallback: number): number { const parsed = Number.parseInt(String(value), 10); @@ -229,6 +230,7 @@ export class ChatAdmissionController { tryAcquireHealthyHeadroom(): ChatAdmissionLease | null { if (this.#activeHealthy >= this.healthyHeadroom) return null; this.#activeHealthy += 1; + const done = trackRequest(); let released = false; return { get released() { @@ -238,6 +240,7 @@ export class ChatAdmissionController { if (released) return; released = true; this.#activeHealthy = Math.max(0, this.#activeHealthy - 1); + done(); }, }; } @@ -264,6 +267,7 @@ export class ChatAdmissionController { tryAcquireHeavy(): ChatAdmissionLease | null { if (this.#activeHeavy >= this.maxHeavyInFlight) return null; this.#activeHeavy += 1; + const done = trackRequest(); let released = false; return { get released() { @@ -273,6 +277,7 @@ export class ChatAdmissionController { if (released) return; released = true; this.#activeHeavy = Math.max(0, this.#activeHeavy - 1); + done(); this.#dispatchFair(); }, }; diff --git a/tests/unit/authz/pipeline.test.ts b/tests/unit/authz/pipeline.test.ts index 34703f270ec..6f6469e804e 100644 --- a/tests/unit/authz/pipeline.test.ts +++ b/tests/unit/authz/pipeline.test.ts @@ -306,6 +306,7 @@ test("runAuthzPipeline rejects new API requests during shutdown drain", async () assert.equal(response.status, 503); assert.equal(body.error.code, "SERVICE_UNAVAILABLE"); + assert.equal(response.headers.get("retry-after"), "5"); }); test("runAuthzPipeline rejects rewritten API aliases during shutdown drain", async () => { @@ -319,6 +320,7 @@ test("runAuthzPipeline rejects rewritten API aliases during shutdown drain", asy assert.equal(response.status, 503); assert.equal(response.headers.get("x-omniroute-route-class"), "CLIENT_API"); assert.equal(body.error.code, "SERVICE_UNAVAILABLE"); + assert.equal(response.headers.get("retry-after"), "5"); }); test("runAuthzPipeline allows dashboard sessions to read model catalog aliases", async () => { diff --git a/tests/unit/chat-body-admission.test.ts b/tests/unit/chat-body-admission.test.ts index e4707402a15..05afae869c8 100644 --- a/tests/unit/chat-body-admission.test.ts +++ b/tests/unit/chat-body-admission.test.ts @@ -16,6 +16,7 @@ const { resolveSelfLoopBearer, } = admissionModule; const { withEarlyStreamKeepalive } = await import("../../open-sse/utils/earlyStreamKeepalive.ts"); +const { getActiveRequestCount } = await import("../../src/lib/gracefulShutdown.ts"); /** * Save/restore the env-var keys that `resolveSelfLoopBearer` reads so tests can @@ -49,6 +50,25 @@ function chatRequest(body: string, contentLength: string | null = String(body.le }); } +test("heavyweight leases are counted for SIGTERM drain (#11015)", () => { + globalThis.__omnirouteShutdown = { init: true, shuttingDown: false, activeRequests: 0 }; + const controller = new ChatAdmissionController(2); + const before = getActiveRequestCount(); + const lease = controller.tryAcquireHeavy(); + assert.ok(lease); + assert.equal(getActiveRequestCount(), before + 1); + const headroom = controller.tryAcquireHealthyHeadroom(); + assert.ok(headroom); + assert.equal(getActiveRequestCount(), before + 2); + lease.release(); + assert.equal(getActiveRequestCount(), before + 1); + headroom.release(); + assert.equal(getActiveRequestCount(), before); + lease.release(); + headroom.release(); + assert.equal(getActiveRequestCount(), before); +}); + test("small known body is admitted without consuming heavyweight capacity", async () => { const controller = new ChatAdmissionController(1); const result = await admitChatRequest(chatRequest("{}"), { From e52d2db44910e1d7825c9bf643e9098d8805ea53 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Sat, 22 Aug 2026 21:53:26 -0400 Subject: [PATCH 14/29] fix(compression): CCR must not strand prompts for callers without the retrieve tool (#11084) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board + this branch: ccr-non-mcp-full-prompt-loss + ccr-retrieval-ramp 20/20, file-size gate green with the ccr/index listing (1024, dated annotation — owner-authorized). The callerSupportsCcrRetrieve gate now skips the whole engine for callers whose tools[] cannot reach omniroute_ccr_retrieve — no more 15KB prompt arriving upstream as 112 tokens. Production-measured root cause, textbook TDD. Thank you @HouMinXi! --- .../services/compression/engines/ccr/index.ts | 26 +++- .../ccr-non-mcp-full-prompt-loss-7746.test.ts | 115 +++++++++++++++--- .../compression/ccr-retrieval-ramp.test.ts | 8 +- 3 files changed, 129 insertions(+), 20 deletions(-) diff --git a/open-sse/services/compression/engines/ccr/index.ts b/open-sse/services/compression/engines/ccr/index.ts index d2136fc8f19..92869bfbc9b 100644 --- a/open-sse/services/compression/engines/ccr/index.ts +++ b/open-sse/services/compression/engines/ccr/index.ts @@ -45,7 +45,7 @@ import { } from "../../../../../src/lib/db/ccrBlocks.ts"; import { createCompressionStats } from "../../stats.ts"; import { queryBlock, type CcrQuery } from "./ccrQuery.ts"; -import { injectCcrProtocolInstruction } from "./protocolInstruction.ts"; +import { callerSupportsCcrRetrieve, injectCcrProtocolInstruction } from "./protocolInstruction.ts"; import type { CompressionEngine, CompressionEngineApplyOptions, @@ -939,6 +939,30 @@ export const ccrEngine: CompressionEngine = { return { body, compressed: false, stats: null }; } + // #7746 follow-up: only callers whose tools[] proves they can reach + // omniroute_ccr_retrieve may have content replaced at all. For everyone + // else (plain OpenAI-compatible clients — the marker is an MCP-only + // contract) replacement would strand the original text behind a hash the + // model has no way to resolve. Skip the whole engine for them. The check + // is wrapped defensively: a malformed body must fail OPEN (no + // compression), never throw into the request pipeline. + let callerCanRetrieve = false; + try { + callerCanRetrieve = callerSupportsCcrRetrieve(body); + } catch (err) { + // Defensive: the helper is total, but if it ever throws we must fail + // OPEN (no compression) — and surface it so a future regression in the + // helper is visible instead of silently bypassing compression forever. + console.warn( + "[compression/ccr] callerSupportsCcrRetrieve threw; skipping compression:", + err instanceof Error ? err.message : err + ); + callerCanRetrieve = false; + } + if (!callerCanRetrieve) { + return { body, compressed: false, stats: null }; + } + const minChars = typeof stepConfig["minChars"] === "number" ? (stepConfig["minChars"] as number) diff --git a/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts b/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts index 0f4abd534e6..a646744871c 100644 --- a/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts +++ b/tests/unit/compression/ccr-non-mcp-full-prompt-loss-7746.test.ts @@ -15,6 +15,7 @@ import { describe, it, before } from "node:test"; import assert from "node:assert/strict"; import { ccrEngine, + getCcrStoreStats, resetCcrStore, retrieveBlock, } from "../../../open-sse/services/compression/engines/ccr/index.ts"; @@ -54,39 +55,117 @@ describe("issue #7746 — CCR must not reduce the sole user prompt to a bare, un }); it("prompt fixture is realistically sized (>= default 600-char minChars)", () => { - assert.ok(REPORTER_PROMPT.length >= 600, `fixture must be >= 600 chars, got ${REPORTER_PROMPT.length}`); + assert.ok( + REPORTER_PROMPT.length >= 600, + `fixture must be >= 600 chars, got ${REPORTER_PROMPT.length}` + ); }); - it("does not leave the model with only the bare CCR marker when no retrieve tool is available", () => { + it("non-MCP caller: CCR skips entirely — the sole user prompt passes through verbatim", () => { resetCcrStore(); const body = makeOpenCodeStyleRequestBody(); const result = ccrEngine.apply(body as Record, { stepConfig: {} }); - assert.equal(result.compressed, true, "CCR compressed the sole user message (reproducing the report)"); - - const messages = result.body.messages as Array<{ role: string; content: string }>; - const compressedContent = messages[0].content; - const isBareMarkerOnly = /^\[CCR retrieve hash=[0-9a-f]{24} chars=\d+\]$/.test(compressedContent); - + // #7746 follow-up (forge review outage, 2026-08-22): the preamble guard was + // not enough — a non-MCP caller received "[CCR retrieve hash=...] markers" + // it had no tool to resolve (upstream saw 112 of ~3.6K tokens). The engine + // now refuses to replace content at all when tools[] lacks + // omniroute_ccr_retrieve: compressed=false, message content untouched. assert.equal( - isBareMarkerOnly, + result.compressed, false, - "BUG #7746: CCR replaced the ENTIRE sole user message with nothing but the bare " + - `[CCR retrieve hash=...] marker, permanently losing the original prompt for any ` + - `non-MCP caller that cannot resolve the marker. Got: ${JSON.stringify(compressedContent)}` + "CCR must not compress for a caller without the retrieve tool" ); + assert.equal(result.stats, null, "no stats when the engine is skipped"); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].role, "user", "message role must stay user"); + assert.equal( + messages[0].content, + REPORTER_PROMPT, + "sole user prompt must pass through verbatim" + ); + assert.equal(messages.length, 1, "no protocol instruction may be injected for non-MCP callers"); + // Guard regression check: if callerSupportsCcrRetrieve ever returned true + // here, the store would silently accumulate blocks no non-MCP caller can + // retrieve. After a skip the store must hold nothing for this principal. + assert.equal(getCcrStoreStats().entries, 0, "store must stay empty after a non-MCP skip"); }); - it("the original prompt remains fully retrievable by hash even after the guard applies", () => { + // tools:[] and unrelated tools are distinct caller shapes that must all be + // treated as non-MCP: an empty array and a foreign tool list both mean the + // retrieve tool is unreachable. + for (const label of ["empty tools array", "unrelated tools"] as const) { + it(`non-MCP caller with ${label}: CCR skips entirely`, () => { + resetCcrStore(); + const tools = + label === "empty tools array" + ? [] + : [ + { type: "function", function: { name: "get_weather" } }, + { type: "function", function: { name: "web_search" } }, + ]; + const body = { ...makeOpenCodeStyleRequestBody(), tools }; + const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + + assert.equal(result.compressed, false, `${label} must not compress`); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].content, REPORTER_PROMPT, "prompt passes through verbatim"); + assert.equal(messages.length, 1, "no protocol instruction injected"); + }); + } + + // A malformed body (tools as a non-array, or entries of unexpected shape) + // must fail OPEN — no compression, never a throw into the request pipeline. + for (const malformed of [ + { tools: "not-an-array" }, + { tools: [null, 42, "x"] }, + { tools: [{}, { type: "function" }] }, + ]) { + it(`malformed tools payload (${JSON.stringify(malformed.tools)}): engine skips without throwing`, () => { + resetCcrStore(); + const body = { ...makeOpenCodeStyleRequestBody(), ...malformed }; + const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + + assert.equal(result.compressed, false, "malformed tools must fail open (skip)"); + const messages = result.body.messages as Array<{ role: string; content: string }>; + assert.equal(messages[0].content, REPORTER_PROMPT, "prompt passes through verbatim"); + assert.equal( + getCcrStoreStats().entries, + 0, + "store must stay empty after a malformed-tools skip" + ); + }); + } + + it("MCP-capable caller (tools[] advertises omniroute_ccr_retrieve): replacement still runs and stays retrievable", () => { resetCcrStore(); - const body = makeOpenCodeStyleRequestBody(); + const body = { + ...makeOpenCodeStyleRequestBody(), + tools: [{ type: "function", function: { name: "omniroute_ccr_retrieve" } }], + }; const result = ccrEngine.apply(body as Record, { stepConfig: {} }); + assert.equal(result.compressed, true, "CCR still compresses for MCP-capable callers"); const messages = result.body.messages as Array<{ role: string; content: string }>; - const compressedContent = messages[0].content; + // The protocol instruction is injected as a leading system message, so the + // compressed conversation is exactly: [instruction, original user message]. + assert.equal(messages.length, 2, "instruction + user message"); + assert.equal(messages[0].role, "system", "instruction is a leading system message"); + assert.ok( + typeof messages[0].content === "string" && messages[0].content.length > 0, + "instruction content must be non-empty" + ); + assert.ok( + messages[0].content.includes("omniroute_ccr_retrieve"), + "instruction must teach the retrieve tool contract" + ); + const compressedContent = messages[1].content; const match = compressedContent.match(/\[CCR retrieve hash=([0-9a-f]{24}) chars=\d+\]/); - assert.ok(match, "compressed content must still contain a resolvable CCR marker"); - const hash = match![1]; - assert.equal(retrieveBlock(hash), REPORTER_PROMPT, "original prompt must be stored verbatim and retrievable"); + assert.ok(match, "compressed content must contain a resolvable CCR marker"); + assert.equal( + retrieveBlock(match![1]), + REPORTER_PROMPT, + "original prompt must be stored verbatim and retrievable" + ); }); }); diff --git a/tests/unit/compression/ccr-retrieval-ramp.test.ts b/tests/unit/compression/ccr-retrieval-ramp.test.ts index 2b475e69bfb..6e87f7b7877 100644 --- a/tests/unit/compression/ccr-retrieval-ramp.test.ts +++ b/tests/unit/compression/ccr-retrieval-ramp.test.ts @@ -78,7 +78,13 @@ describe("ccrEngine.apply — retrieval-aware compression (H8)", () => { const block = (len: number) => "x".repeat(len); const run = (content: string, retrievalRampFactor = 2) => ccrEngine.apply( - { messages: [{ role: "user", content }] }, + // The retrieve tool is advertised — this suite exercises the compression + // path itself (H8 ramp); without the tool declaration the #7746 guard + // skips the engine entirely. + { + messages: [{ role: "user", content }], + tools: [{ type: "function", function: { name: "omniroute_ccr_retrieve" } }], + }, { stepConfig: { minChars: BASE, retrievalRampFactor }, principalId: P } ); From c018bb41a7eade34b56fa9bc7e937ae962a745b9 Mon Sep 17 00:00:00 2001 From: Xiangzhe <32761048+xz-dev@users.noreply.github.com> Date: Sun, 23 Aug 2026 09:54:49 +0800 Subject: [PATCH 15/29] fix(catalog): declare GLM reasoning effort tiers (#10963) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged after conflict resolution in modelMetadataRegistry.ts: the tip's effortTiers chain (declared efforts → declared tiers → undefined-if-thinking-declared → codex extension) now carries this PR's GLM guard as the final-fallback override — GLM-family models without a provider-declared contract get the authoritative empty tier list instead of generic OpenAI tiers. GLM/ZCode suites 40/40 on the resolved branch. Closes #10962. Thank you @xz-dev! --- docs/changelog/fragments/10962.md | 1 + open-sse/config/glmProvider.ts | 29 +++++- .../config/providers/registry/zcode/index.ts | 15 ++- open-sse/executors/glm.ts | 2 +- open-sse/executors/zcode.ts | 6 +- .../__tests__/glmCodingProviderConfig.test.ts | 23 +++++ open-sse/utils/syncedEffortVariants.ts | 16 ++-- src/lib/modelMetadataRegistry.ts | 24 +++-- .../glm-5.3-catalog-and-effort-tiers.test.ts | 92 ++++++++++++++++++- tests/unit/zcode-executor.test.ts | 4 +- tests/unit/zcode-provider.test.ts | 15 ++- 11 files changed, 196 insertions(+), 31 deletions(-) create mode 100644 docs/changelog/fragments/10962.md diff --git a/docs/changelog/fragments/10962.md b/docs/changelog/fragments/10962.md new file mode 100644 index 00000000000..5170a415e3f --- /dev/null +++ b/docs/changelog/fragments/10962.md @@ -0,0 +1 @@ +fix(catalog): expose only provider-routable GLM reasoning-effort tiers and remove unroutable ZCode aliases diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 9c1580e7aed..8668de2c636 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -19,17 +19,16 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ export const GLM_SHARED_MODELS = Object.freeze([ { - // GLM-5.3 (2026-08-14): one upstream id; effort is the reasoning_effort - // param (low|high|max, default max) — the -high/-low entries below are - // OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. - // Default context window not yet published by Z.ai; 1M mirrored from - // GLM-5.2 (same base model). https://z.ai/blog/glm-5.3 + // GLM-5.3 exposes low|high|max reasoning_effort (default max); -high/-low + // are OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. + // https://docs.z.ai/guides/llm/glm-5.3 id: "glm-5.3", name: "GLM 5.3", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], }, { id: "glm-5.3-high", @@ -38,6 +37,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.3-low", @@ -46,14 +46,19 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low"], }, { + // GLM-5.2 has two positive effective tiers: low/medium map to high and xhigh + // maps to max; disabling thinking remains the separate thinking toggle. + // https://docs.z.ai/guides/capabilities/thinking id: "glm-5.2", name: "GLM 5.2", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high", "max"], }, { id: "glm-5.2-high", @@ -62,6 +67,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.2-max", @@ -70,14 +76,18 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["max"], }, { + // Earlier GLM families support the thinking toggle, not reasoning_effort. + // An explicit empty list prevents generic catalog tiers from being inferred. id: "glm-5.1", name: "GLM 5.1", contextLength: 204800, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5", @@ -86,6 +96,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5-turbo", @@ -94,6 +105,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7-flash", @@ -102,6 +114,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7", @@ -110,6 +123,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.6v", @@ -118,6 +132,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -127,6 +142,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5v", @@ -135,6 +151,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -144,6 +161,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5-air", @@ -152,6 +170,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, ]); diff --git a/open-sse/config/providers/registry/zcode/index.ts b/open-sse/config/providers/registry/zcode/index.ts index cd2a4eece64..65e2e1c3d3d 100644 --- a/open-sse/config/providers/registry/zcode/index.ts +++ b/open-sse/config/providers/registry/zcode/index.ts @@ -1,6 +1,17 @@ import type { RegistryEntry } from "../../shared.ts"; import { GLM_SHARED_MODELS } from "../../../glmProvider.ts"; +const GLM_EXECUTOR_EFFORT_ALIASES = new Set([ + "glm-5.3-high", + "glm-5.3-low", + "glm-5.2-high", + "glm-5.2-max", +]); + +export const ZCODE_MODELS = GLM_SHARED_MODELS.filter( + (model) => !GLM_EXECUTOR_EFFORT_ALIASES.has(model.id) +).map((model) => ({ ...model, supportedThinkingEfforts: [] })); + /** * Local ZCode app-server backend. Authentication remains in the user's local * ZCode profile (`builtin:zai-coding-plan`); OmniRoute does not receive or @@ -14,5 +25,7 @@ export const zcodeProvider: RegistryEntry = { baseUrl: "zcode://app-server/stdio", authType: "none", authHeader: "none", - models: [...GLM_SHARED_MODELS], + // ZCode's app-server transport does not consume reasoning_effort; keep thinking + // capability metadata without advertising aliases or tiers that it would ignore. + models: ZCODE_MODELS, }; diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index a222571022a..59e18719076 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -73,7 +73,7 @@ type GlmEffortTier = { * `thinking.type=enabled` (5.3 no longer accepts thinking disabled). * * https://docs.z.ai/devpack/latest-model - * https://z.ai/blog/glm-5.3 + * https://docs.z.ai/guides/llm/glm-5.3 */ function parseGlmEffortTier(model: string): GlmEffortTier | null { switch (model) { diff --git a/open-sse/executors/zcode.ts b/open-sse/executors/zcode.ts index 0841b4daa8e..8f0a98ac142 100644 --- a/open-sse/executors/zcode.ts +++ b/open-sse/executors/zcode.ts @@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto"; import { existsSync } from "node:fs"; import { homedir } from "node:os"; import { join, resolve } from "node:path"; -import { GLM_SHARED_MODELS } from "../config/glmProvider.ts"; +import { ZCODE_MODELS } from "../config/providers/registry/zcode/index.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorExecuteResult, type ProviderCredentials } from "./base.ts"; import { ZcodeAppServerClient, type ZcodeClientLike } from "./zcodeProtocol.ts"; import { buildErrorBody, errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; @@ -12,8 +12,8 @@ const DEFAULT_PROVIDER_ID = "builtin:zai-coding-plan"; const DEFAULT_TURN_TIMEOUT_MS = 120_000; const DEFAULT_POLL_INTERVAL_MS = 250; const TERMINAL_STATUSES = new Set(["completed", "idle", "paused", "error"]); -const ZCODE_MODEL_ALLOWLIST = new Set(GLM_SHARED_MODELS.map((model) => model.id)); -const DEFAULT_ZCODE_MODEL = GLM_SHARED_MODELS[0]?.id || "glm-5.2"; +const ZCODE_MODEL_ALLOWLIST = new Set(ZCODE_MODELS.map((model) => model.id)); +const DEFAULT_ZCODE_MODEL = ZCODE_MODELS[0]?.id || "glm-5.2"; type JsonRecord = Record; type OpenAIMsg = { role?: string; content?: unknown }; diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index 8b7f3cee324..50c16fa9665 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -106,6 +106,29 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("declares exact GLM reasoning-effort tiers across every shared GLM provider", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt"]) { + for (const model of getModelsByProviderId(provider)) { + expect(model.supportedThinkingEfforts, `${provider}/${model.id} effort tiers`).toEqual( + routedTiers.get(model.id) ?? [] + ); + } + } + + for (const model of getModelsByProviderId("zcode")) { + expect(model.supportedThinkingEfforts, `zcode/${model.id} effort tiers`).toEqual([]); + } + }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); diff --git a/open-sse/utils/syncedEffortVariants.ts b/open-sse/utils/syncedEffortVariants.ts index 2a2c7f0d68c..33c5ec8c56b 100644 --- a/open-sse/utils/syncedEffortVariants.ts +++ b/open-sse/utils/syncedEffortVariants.ts @@ -19,17 +19,17 @@ * only when the base model's own `supportedThinkingEfforts` actually declares that tier — * never a blind string match. * - * Skipped entirely for `codex` and `kimi`-owned models: both already own a conflicting - * native `-{effort}` suffix mechanism (`splitCodexReasoningSuffix` / - * `getKimiCodeStaticThinkingPolicy`), so double-registering here would collide with their - * own alias resolution. Also skipped for any model whose id already ends in a token that - * matches a canonical effort value, to avoid colliding with a model that legitimately ends - * in an effort-like token (e.g. a model literally named "...-high"). + * Skipped entirely for `codex`, `kimi`-owned, and GLM (`glm`, `glm-cn`, `glmt`) models: + * they already own conflicting `-{effort}` aliases (`splitCodexReasoningSuffix`, + * `getKimiCodeStaticThinkingPolicy`, or `GlmExecutor::parseGlmEffortTier`), so generating + * another layer here would create invalid nested ids. Also skipped for any model whose id + * already ends in a token that matches a canonical effort value, to avoid colliding with a + * model that legitimately ends in an effort-like token (e.g. a model named "...-high"). */ import { CANONICAL_EFFORT_VALUES } from "@/shared/reasoning/effortStandardization.ts"; -/** Provider ids that already own a native `-{effort}` suffix mechanism — never double-register. */ -export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex"]); +/** Provider ids with dedicated `-{effort}` aliases — never synthesize another suffix layer. */ +export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex", "glm", "glm-cn", "glmt"]); /** Provider-id prefixes covering that mechanism's multiple connection variants (kimi-coding, kimi-coding-apikey). */ const SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES = ["kimi"]; diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 3828047b90b..b7aa2aa2164 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -130,6 +130,11 @@ function uniqueStrings(values: Array) { ]; } +export function isGlmFamilyModel(modelId: string, displayName = ""): boolean { + const glmFamilyPattern = /(?:^|[/@:_. -])glm(?=$|[-._ /@:](?:z)?\d|\d)/i; + return glmFamilyPattern.test(modelId) || glmFamilyPattern.test(displayName); +} + function toQualifiedId( providerAlias: string | null, provider: string | null, @@ -477,11 +482,16 @@ export function enrichCatalogModelEntry( ? declaredEffortTiers : sourceDeclaresThinking ? undefined - : extendCodexGpt56EffortValues( - metadata.provider, - metadata.model, - CANONICAL_EFFORT_VALUES - ); + : // #10963: GLM-family models never inherit generic OpenAI tiers — an + // explicit empty list is authoritative unless a provider-declared + // contract exists (handled by declaredEffortTiers above). + isGlmFamilyModel(metadata.model, metadata.displayName) + ? [] + : extendCodexGpt56EffortValues( + metadata.provider, + metadata.model, + CANONICAL_EFFORT_VALUES + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -502,7 +512,9 @@ export function enrichCatalogModelEntry( // #6241: surface thinking support + the canonical effort tiers so the frontend can // render the effort/thinking toggles. `thinking` is kept for back-compat; `supportsThinking` // is the explicit flag and `effort_tiers` lists the selectable reasoning levels - // (only when the model actually supports thinking). + // (only when the model actually supports thinking). An explicit empty registry list + // is authoritative; GLM models also require a provider-declared contract instead of + // inheriting generic OpenAI effort tiers. ...(typeof metadata.capabilities.supportsThinking === "boolean" ? { thinking: metadata.capabilities.supportsThinking, diff --git a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts index d14c136e7b3..5d927de02a8 100644 --- a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts +++ b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts @@ -1,7 +1,7 @@ import test from "node:test"; import assert from "node:assert/strict"; -// GLM-5.3 support (released 2026-08-14, https://z.ai/blog/glm-5.3). +// GLM-5.3 support (released 2026-08-14, https://docs.z.ai/guides/llm/glm-5.3). // // Upstream ships ONE model id (`glm-5.3`) — effort is a request parameter // (`reasoning_effort`: low|high|max, default max) on the coding chat/completions @@ -12,14 +12,15 @@ import assert from "node:assert/strict"; // beta header), the 5.3 tiers use the documented `reasoning_effort` param on the // OpenAI coding transport. // -// Spec caveat: Z.ai has not yet published the default context window — 1M is -// mirrored from GLM-5.2 (same base model) per operator decision; correct when -// the official spec lands. +// Z.AI documents a 1M context window and 128K maximum output. -const { getRegistryEntry } = await import("../../open-sse/config/providerRegistry.ts"); +const { getRegistryEntry, REGISTRY } = await import("../../open-sse/config/providerRegistry.ts"); const { GlmExecutor } = await import("../../open-sse/executors/glm.ts"); const { MODEL_SPECS } = await import("../../src/shared/constants/modelSpecs.ts"); const { GLM_PRICING } = await import("../../src/shared/constants/pricing/shared-tiers.ts"); +const metadataRegistry = await import("../../src/lib/modelMetadataRegistry.ts"); +const { shouldExposeSyncedEffortVariants, SYNCED_EFFORT_SKIP_PROVIDERS } = + await import("../../open-sse/utils/syncedEffortVariants.ts"); const GLM_5_3_IDS = ["glm-5.3", "glm-5.3-high", "glm-5.3-low"] as const; @@ -38,6 +39,87 @@ function modelIds(provider: string): string[] { return (entry.models ?? []).map((m) => m.id); } +test("shared GLM providers keep their dedicated aliases instead of synthesizing another layer", () => { + for (const provider of ["glm", "glm-cn", "glmt"]) { + assert.ok(SYNCED_EFFORT_SKIP_PROVIDERS.has(provider), provider); + assert.equal( + shouldExposeSyncedEffortVariants({ + id: `${provider}/glm-5.3`, + owned_by: provider, + capabilities: { effort_tiers: ["low", "high", "max"] }, + }), + false, + provider + ); + } + assert.equal(SYNCED_EFFORT_SKIP_PROVIDERS.has("zcode"), false); +}); + +test("GLM family detection covers numeric, Z1, and bare provider model ids", () => { + for (const modelId of [ + "hf:zai-org/GLM-5.2", + "THUDM/GLM-Z1-32B-0414", + "THUDM/GLM-Z1-9B-0414", + "glm", + ]) { + assert.equal(metadataRegistry.isGlmFamilyModel(modelId), true, modelId); + } + assert.equal(metadataRegistry.isGlmFamilyModel("llama-3.3"), false); +}); + +test("catalog suppresses inferred tiers for every GLM registry entry without a provider contract", () => { + let audited = 0; + for (const [provider, entry] of Object.entries(REGISTRY)) { + for (const model of entry.models ?? []) { + if (!metadataRegistry.isGlmFamilyModel(model.id, model.name)) continue; + audited += 1; + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + if (capabilities.supportsThinking === true) { + assert.deepEqual( + capabilities.effort_tiers, + model.supportedThinkingEfforts ?? [], + `${provider}/${model.id}` + ); + } else { + assert.equal("effort_tiers" in capabilities, false, `${provider}/${model.id}`); + } + } + } + assert.ok(audited > 0); +}); + +test("catalog exposes only GLM effort tiers that each provider can route", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt", "zcode"]) { + for (const model of getRegistryEntry(provider)!.models ?? []) { + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + const expected = provider === "zcode" ? [] : (routedTiers.get(model.id) ?? []); + assert.equal(capabilities.supportsThinking, true, `${provider}/${model.id}`); + assert.deepEqual(capabilities.effort_tiers, expected, `${provider}/${model.id}`); + } + } +}); + for (const provider of ["glm", "glm-cn", "glmt"]) { test(`${provider} advertises the GLM-5.3 base model and effort tiers (GLM_SHARED_MODELS)`, () => { const ids = modelIds(provider); diff --git a/tests/unit/zcode-executor.test.ts b/tests/unit/zcode-executor.test.ts index c82e7b33683..db4a8586bee 100644 --- a/tests/unit/zcode-executor.test.ts +++ b/tests/unit/zcode-executor.test.ts @@ -26,6 +26,8 @@ function requestBody() { test("ZCode accepts GLM Coding Plan models and rejects unsafe/unknown ids", async () => { const { resolveZcodeModel } = await loadZcodeExecutor(); assert.deepEqual(resolveZcodeModel("glm-5.2"), { ok: true, model: "glm-5.2" }); + assert.equal(resolveZcodeModel("glm-5.2-high").ok, false); + assert.equal(resolveZcodeModel("glm-5.3-low").ok, false); assert.equal(resolveZcodeModel("-unexpected").ok, false); assert.equal(resolveZcodeModel("unknown-model").ok, false); }); @@ -70,7 +72,7 @@ test("ZCode buffers the completed turn into OpenAI SSE when stream=true", async }); const result = await executor.execute({ - model: "glm-5.2-high", + model: "glm-5.2", body: requestBody(), stream: true, credentials: {}, diff --git a/tests/unit/zcode-provider.test.ts b/tests/unit/zcode-provider.test.ts index 3acf3c862c4..e882ab5fd91 100644 --- a/tests/unit/zcode-provider.test.ts +++ b/tests/unit/zcode-provider.test.ts @@ -10,5 +10,18 @@ test("ZCode provider registry exposes a local no-auth GLM Coding Plan backend", assert.equal(zcodeProvider.baseUrl, "zcode://app-server/stdio"); assert.equal(zcodeProvider.authType, "none"); assert.equal(zcodeProvider.authHeader, "none"); - assert.equal(zcodeProvider.models.some((model) => model.id === "glm-5.2"), true); + assert.equal( + zcodeProvider.models.some((model) => model.id === "glm-5.2"), + true + ); + for (const alias of ["glm-5.3-high", "glm-5.3-low", "glm-5.2-high", "glm-5.2-max"]) { + assert.equal( + zcodeProvider.models.some((model) => model.id === alias), + false, + alias + ); + } + for (const model of zcodeProvider.models) { + assert.deepEqual(model.supportedThinkingEfforts, [], model.id); + } }); From 9689dcef9fc500d8dad91d78b2b95401aace4bb4 Mon Sep 17 00:00:00 2001 From: Ke Jin Date: Sun, 23 Aug 2026 09:58:58 +0800 Subject: [PATCH 16/29] fix(reasoning): preserve mixed plaintext and drop incompatible state (#10949, #10959) (#10961) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board + this branch: 231/231 across chatcore-translation-paths, reasoning-cache, strip-reasoning-blobs, and both Responses translator suites. Pre-merge: propagated the #11110/#11129 summary:[] defaults into five assertions here (each commented with its PR) — without it this branch red against the tip, and as a bonus the merge drains the 4 reasoning reds that were live on the tip from those merges. Plaintext now wins over a coexisting opaque companion; opaque-only drops cleanly for plaintext targets; combos keep explicit Skip. Fixes #10949 and #10959. Thank you @jackjinke! --- .../10550-responses-reasoning-transport.md | 2 +- .../fixes/10949-mixed-reasoning-plaintext.md | 1 + open-sse/executors/codex.ts | 1 - open-sse/handlers/chatCore.ts | 4 +- open-sse/services/reasoningInputPolicy.ts | 48 ++++---- .../translator/request/openai-responses.ts | 12 +- src/app/(dashboard)/dashboard/combos/page.tsx | 9 -- src/sse/handlers/chat.ts | 2 +- src/sse/handlers/chatHelpers.ts | 2 +- tests/unit/chatcore-translation-paths.test.ts | 49 ++++---- tests/unit/reasoning-cache.test.ts | 105 +++++++++++------- ...asoning-blobs-agentic-context-1599.test.ts | 60 ++++++++-- .../translator-openai-responses-req.test.ts | 38 +++---- .../translator-resp-openai-responses.test.ts | 46 ++++++++ 14 files changed, 234 insertions(+), 145 deletions(-) create mode 100644 changelog.d/fixes/10949-mixed-reasoning-plaintext.md diff --git a/changelog.d/fixes/10550-responses-reasoning-transport.md b/changelog.d/fixes/10550-responses-reasoning-transport.md index d34c433debd..e2b40cdb8c9 100644 --- a/changelog.d/fixes/10550-responses-reasoning-transport.md +++ b/changelog.d/fixes/10550-responses-reasoning-transport.md @@ -1 +1 @@ -- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Combos now drop incompatible continuation reasoning by default and can explicitly skip incompatible targets, while known providers no longer show redundant encrypted-reasoning controls. (#10550) +- Preserve portable plaintext reasoning by default across streaming and non-streaming Chat Completions and Responses routes while keeping provider-bound opaque state target-compatible. Direct requests drop incompatible continuation reasoning by default; combos can explicitly skip incompatible targets without mutating the request. Known providers no longer show redundant encrypted-reasoning controls. (#10550, #10959) diff --git a/changelog.d/fixes/10949-mixed-reasoning-plaintext.md b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md new file mode 100644 index 00000000000..05a055ec538 --- /dev/null +++ b/changelog.d/fixes/10949-mixed-reasoning-plaintext.md @@ -0,0 +1 @@ +- Preserve explicit plaintext reasoning when a Responses reasoning item also carries opaque provider state (rare OpenCode Go `deepseek-v4-flash` responses). Mixed plaintext + opaque input is projected onto the target transport: plaintext targets keep portable text, opaque targets keep provider state. Opaque-only reasoning is dropped when the selected target cannot replay it, allowing cross-model conversations to continue. (#10949, #10959) diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index 6e39bb81caa..b9f6cf113a2 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -1398,7 +1398,6 @@ export class CodexExecutor extends BaseExecutor { provider: "codex", preserveEncryptedReasoning: credentials?.providerSpecificData?.preserveEncryptedReasoning === true, - onIncompatibleReasoning: "drop", }); if (nativeCodexPassthrough) { diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 8f11373b1fb..fea8640f9d3 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -516,7 +516,7 @@ export async function handleChatCore({ conversationId = null, modelPinned = false, skipResourcePressureGuard = false, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", managedLease = null, }) { let { provider, model, extendedContext } = modelInfo; @@ -1213,7 +1213,7 @@ export async function handleChatCore({ provider, preserveEncryptedReasoning: credentials?.providerSpecificData?.preserveEncryptedReasoning === true, - onIncompatibleReasoning: reasoningTransportFallback === "drop" ? "drop" : "reject", + onIncompatibleReasoning: reasoningTransportFallback === "skip" ? "reject" : "drop", } ); if (policy.incompatibleReasoning) { diff --git a/open-sse/services/reasoningInputPolicy.ts b/open-sse/services/reasoningInputPolicy.ts index 94f27ddcdc9..0e9e8d194db 100644 --- a/open-sse/services/reasoningInputPolicy.ts +++ b/open-sse/services/reasoningInputPolicy.ts @@ -37,7 +37,6 @@ export interface ReasoningInputPolicyOptions { export interface ReasoningInputPolicyResult { incompatibleReasoning: boolean; } - export function resolveReasoningTransport( provider: string | null | undefined, preserveEncryptedReasoning = false @@ -47,18 +46,6 @@ export function resolveReasoningTransport( return transport ?? (preserveEncryptedReasoning ? "opaque" : "plaintext"); } -export function createReasoningTransportIncompatibleError(): Error & { - statusCode: number; - errorType: string; -} { - const error = new Error( - "Reasoning continuation is not compatible with the selected target" - ) as Error & { statusCode: number; errorType: string }; - error.statusCode = 400; - error.errorType = "reasoning_transport_incompatible"; - return error; -} - function asRecord(value: unknown): JsonRecord | null { return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null; } @@ -101,11 +88,12 @@ function hasChatPlaintextReasoning(record: JsonRecord): boolean { /** * Returns only provider-authentic plaintext continuation state. Display summaries - * are excluded, and a record carrying opaque state is never cross-converted. + * and opaque-only records are excluded. Explicit plaintext remains independently + * portable when the same record also carries an opaque companion (#10949). */ export function extractReplayableResponsesReasoningText(value: unknown): string { const record = asRecord(value); - if (!record || record.type !== "reasoning" || hasOpaqueReasoningState(record)) return ""; + if (!record || record.type !== "reasoning") return ""; if (!Array.isArray(record.content)) return ""; return record.content @@ -307,9 +295,9 @@ function sanitizeResponsesInput( } /** - * Applies one protocol-independent compatibility decision before request translation. - * Plaintext is portable by default; opaque state requires an explicit target declaration. - * Display summaries do not affect compatibility; stateless input drops orphan summaries. + * Projects reasoning continuation onto the selected target transport. + * Incompatible active state is dropped by default; combo routing may reject an + * attempt instead so it can fall through without mutating the request. */ export function applyReasoningInputPolicy( body: Record, @@ -321,14 +309,17 @@ export function applyReasoningInputPolicy( inputFormat === "responses" ? inspectResponsesReasoning(body.input) : inspectChatReasoning(body.messages); - const incompatibleReasoning = !isReasoningCompatible(inspection, transport); + const mixedState = inspection.hasPlaintext && inspection.hasOpaque; + const incompatibleReasoning = !mixedState && !isReasoningCompatible(inspection, transport); + // Mixed plaintext + opaque input (#10949) is never a rejection: it is projected + // onto the target transport by the per-item sanitizers below. - if (incompatibleReasoning && options.onIncompatibleReasoning !== "drop") { + if (incompatibleReasoning && options.onIncompatibleReasoning === "reject") { return { incompatibleReasoning: true }; } if (inputFormat === "chat") { - if (incompatibleReasoning && Array.isArray(body.messages)) { + if ((incompatibleReasoning || mixedState) && Array.isArray(body.messages)) { body.messages = dropIncompatibleChatReasoning(body.messages, transport); } return { incompatibleReasoning: false }; @@ -343,12 +334,13 @@ export function applyReasoningInputPolicy( }, ]; } - if (!Array.isArray(body.input)) return { incompatibleReasoning: false }; - body.input = sanitizeResponsesInput( - body.input, - transport, - incompatibleReasoning, - body.store === false - ); + if (Array.isArray(body.input)) { + body.input = sanitizeResponsesInput( + body.input, + transport, + incompatibleReasoning || mixedState, + body.store === false + ); + } return { incompatibleReasoning: false }; } diff --git a/open-sse/translator/request/openai-responses.ts b/open-sse/translator/request/openai-responses.ts index b8866502524..d48dcdf1816 100644 --- a/open-sse/translator/request/openai-responses.ts +++ b/open-sse/translator/request/openai-responses.ts @@ -8,11 +8,7 @@ import { isOpenAIResponsesStoreEnabled } from "@/lib/providers/requestDefaults"; import { FORMATS } from "../formats.ts"; import { register } from "../registry.ts"; import { normalizeResponsesInputForChat } from "../../utils/responsesInputNormalization.ts"; -import { - createReasoningTransportIncompatibleError, - hasOpaqueReasoningState, - extractReplayableResponsesReasoningText, -} from "../../services/reasoningInputPolicy.ts"; +import { extractReplayableResponsesReasoningText } from "../../services/reasoningInputPolicy.ts"; import { getRegisteredProviders, requiresPlainStringContent, @@ -454,10 +450,8 @@ export function openaiResponsesToOpenAIRequest( if (itemType === "reasoning") { // Only genuine plaintext reasoning can cross into Chat reasoning_content. - // Opaque encrypted state and its display summary have no Chat replay form. - if (preserveReasoningContent && hasOpaqueReasoningState(item)) { - throw createReasoningTransportIncompatibleError(); - } + // Opaque encrypted state and its display summary have no Chat replay form, + // so opaque-only items are dropped while mixed items replay their plaintext. if (preserveReasoningContent) { const reasoning = extractReplayableResponsesReasoningText(item); if (reasoning) { diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 7cb1c6c0eea..7ca3af20248 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -3896,15 +3896,6 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders, combo )} - {config.reasoningTransportFallback !== "skip" && ( -

- {getI18nOrFallback( - t, - "reasoningTransportFallbackDropWarning", - "May lose continuation context or cause tool-call continuations to fail." - )} -

- )}
{ diff --git a/tests/unit/chatcore-translation-paths.test.ts b/tests/unit/chatcore-translation-paths.test.ts index 08cc564cc4e..6ecb0487889 100644 --- a/tests/unit/chatcore-translation-paths.test.ts +++ b/tests/unit/chatcore-translation-paths.test.ts @@ -369,7 +369,7 @@ async function invokeChatCore({ onCredentialsRefreshed = null, onRequestSuccess = null, sessionAffinityKey = null, - reasoningTransportFallback = "skip", + reasoningTransportFallback = "drop", managedLease = null, cachedSettings = null, }: any = {}) { @@ -631,7 +631,7 @@ test("chatCore translates a streaming Responses upstream for a Chat client", asy assert.match(streamed, /"content":"ok"/); assert.match(streamed, /data: \[DONE\]/); }); -test("chatCore rejects opaque reasoning for unknown Responses targets unless explicitly enabled", async () => { +test("chatCore drops opaque reasoning for plaintext Responses targets by default (#10959)", async () => { const reasoningItems = [ { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }, { type: "reasoning", encrypted_content: "" }, @@ -640,7 +640,7 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp { id: "fc_call", type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, ]; - const rejected = await invokeChatCore({ + const dropped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/responses", @@ -656,9 +656,12 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp responseFormat: "openai-responses", }); - assert.equal(rejected.result.success, false); - assert.equal(rejected.result.status, 400); - assert.equal(rejected.calls.length, 0); + assert.equal(dropped.result.success, true); + assert.equal(dropped.calls.length, 1); + assert.deepEqual( + dropped.call.body.input.filter((item) => item.type === "reasoning"), + [{ type: "reasoning", summary: [{ text: "not self-contained" }] }] + ); const enabled = await invokeChatCore({ provider: "openai-compatible-sp-openai", @@ -682,7 +685,7 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp assert.deepEqual( input.filter((item) => item.type === "reasoning"), [ - { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }, + { id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }, // summary defaulted by #11110 { type: "reasoning", summary: [{ text: "not self-contained" }] }, ] ); @@ -693,9 +696,9 @@ test("chatCore rejects opaque reasoning for unknown Responses targets unless exp assert.equal(input.find((item) => item.type === "function_call")?.id, undefined); }); -test("chatCore applies Chat reasoning compatibility before stream mode diverges", async () => { +test("chatCore drops incompatible Chat reasoning before stream mode diverges (#10959)", async () => { for (const stream of [false, true]) { - const rejected = await invokeChatCore({ + const dropped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/chat/completions", @@ -728,14 +731,14 @@ test("chatCore applies Chat reasoning compatibility before stream mode diverges" }, }); - assert.equal(rejected.result.success, false, `stream=${stream}`); - assert.equal(rejected.result.status, 400, `stream=${stream}`); - assert.equal(rejected.calls.length, 0, `stream=${stream}`); + assert.equal(dropped.result.success, true, `stream=${stream}`); + assert.equal(dropped.calls.length, 1, `stream=${stream}`); + assert.equal(dropped.call.body.messages[0].reasoning_details, undefined, `stream=${stream}`); } }); -test("chatCore can drop incompatible reasoning for an opted-in Combo attempt", async () => { - const dropped = await invokeChatCore({ +test("chatCore preserves Combo skip behavior for incompatible reasoning", async () => { + const skipped = await invokeChatCore({ provider: "openai-compatible-sp-openai", model: "gpt-5.4", endpoint: "/v1/responses", @@ -757,15 +760,12 @@ test("chatCore can drop incompatible reasoning for an opted-in Combo attempt", a }, responseFormat: "openai-responses", isCombo: true, - reasoningTransportFallback: "drop", + reasoningTransportFallback: "skip", }); - assert.equal(dropped.result.success, true); - assert.equal(dropped.calls.length, 1); - assert.equal( - dropped.call.body.input.some((item) => item.type === "reasoning"), - false - ); + assert.equal(skipped.result.success, false); + assert.equal(skipped.result.status, 400); + assert.equal(skipped.calls.length, 0); }); test("chatCore carries Chat reasoning_content into official DeepSeek Responses input", async () => { @@ -800,6 +800,7 @@ test("chatCore carries Chat reasoning_content into official DeepSeek Responses i assert.deepEqual(call.body.input.slice(0, 3), [ { type: "reasoning", + summary: [], // defaulted on freshly-built reasoning items (#11129) content: [{ type: "reasoning_text", text: "Inspect before calling the tool" }], }, { @@ -866,7 +867,7 @@ test("chatCore replays nonstream DeepSeek Responses reasoning across a Chat tool assert.equal(second.result.success, true); assert.deepEqual( second.call.body.input.find((item) => item.type === "reasoning"), - { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }] } + { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], summary: [] } // summary defaulted by #11129 ); }); @@ -930,7 +931,7 @@ test("chatCore replays streamed DeepSeek Responses reasoning across a Chat tool assert.equal(second.result.success, true); assert.deepEqual( second.call.body.input.find((item) => item.type === "reasoning"), - { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }] } + { type: "reasoning", content: [{ type: "reasoning_text", text: reasoning }], summary: [] } // summary defaulted by #11129 ); }); @@ -1132,7 +1133,7 @@ test("chatCore automatically preserves provider-generated opaque reasoning for C assert.equal(result.success, true); assert.deepEqual( call.body.input.filter((item) => item.type === "reasoning"), - [{ id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob" }] + [{ id: "rs_valid", type: "reasoning", encrypted_content: "encrypted-blob", summary: [] }] // summary defaulted by #11110 ); assert.equal( call.body.input.some((item) => item.type === "item_reference"), diff --git a/tests/unit/reasoning-cache.test.ts b/tests/unit/reasoning-cache.test.ts index aded469ca85..db538a80314 100644 --- a/tests/unit/reasoning-cache.test.ts +++ b/tests/unit/reasoning-cache.test.ts @@ -752,47 +752,70 @@ describe("Reasoning Replay Cache — Translator Replay", () => { assert.equal(lookupReasoning(callId), "Authentic provider reasoning"); }); - it("should never cache Responses summaries or opaque plaintext companions", () => { - for (const [suffix, reasoningItem] of [ - [ - "summary", - { - type: "reasoning", - summary: [{ type: "summary_text", text: "Display-only summary" }], - }, - ], - [ - "mixed", - { - type: "reasoning", - encrypted_content: "opaque-provider-state", - content: [{ type: "reasoning_text", text: "Unsafe plaintext companion" }], - summary: [{ type: "summary_text", text: "Display-only mixed summary" }], - }, - ], - ] as const) { - clearReasoningCacheAll(); - const callId = `call_nonstream_${suffix}_reasoning`; - const translated = translateNonStreamingResponse( - { - object: "response", - model: "deepseek-v4-flash", - output: [ - reasoningItem, - { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, - ], - }, - FORMATS.OPENAI_RESPONSES, - FORMATS.OPENAI - ) as { choices?: Array<{ message?: Record }> }; - const message = translated.choices?.[0]?.message; - - assert.ok(message); - assert.equal(message.reasoning_content, undefined); - assert.ok(Array.isArray(message.reasoning_summary)); - assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 0); - assert.equal(lookupReasoning(callId), null); - } + it("preserves plaintext reasoning from a mixed plaintext + encrypted_content item (#10949)", () => { + clearReasoningCacheAll(); + const callId = "call_nonstream_mixed_reasoning"; + const translated = translateNonStreamingResponse( + { + object: "response", + model: "deepseek-v4-flash", + output: [ + { + type: "reasoning", + content: [ + { + type: "reasoning_text", + text: "Let me start by reading the directory to understand the structure of the corpus.", + }, + ], + encrypted_content: "", + summary: [], + }, + { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, + ], + }, + FORMATS.OPENAI_RESPONSES, + FORMATS.OPENAI + ) as { choices?: Array<{ message?: Record }> }; + const message = translated.choices?.[0]?.message; + + assert.ok(message); + assert.equal( + message.reasoning_content, + "Let me start by reading the directory to understand the structure of the corpus." + ); + assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 1); + assert.equal( + lookupReasoning(callId), + "Let me start by reading the directory to understand the structure of the corpus." + ); + }); + + it("should never cache summary-only Responses reasoning", () => { + clearReasoningCacheAll(); + const callId = "call_nonstream_summary_reasoning"; + const translated = translateNonStreamingResponse( + { + object: "response", + model: "deepseek-v4-flash", + output: [ + { + type: "reasoning", + summary: [{ type: "summary_text", text: "Display-only summary" }], + }, + { type: "function_call", call_id: callId, name: "read_file", arguments: "{}" }, + ], + }, + FORMATS.OPENAI_RESPONSES, + FORMATS.OPENAI + ) as { choices?: Array<{ message?: Record }> }; + const message = translated.choices?.[0]?.message; + + assert.ok(message); + assert.equal(message.reasoning_content, undefined); + assert.ok(Array.isArray(message.reasoning_summary)); + assert.equal(cacheReasoningFromAssistantMessage(message, "deepseek", "deepseek-v4-flash"), 0); + assert.equal(lookupReasoning(callId), null); }); it("should preserve client-provided reasoning content", () => { diff --git a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts index 25eae7a839c..8b3059236dd 100644 --- a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts +++ b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts @@ -8,7 +8,7 @@ import { omitEncryptedReasoningForLog } from "../../src/lib/logPayloads.ts"; // Responses reasoning replay is target-scoped. Plaintext DeepSeek state and // provider-generated opaque state are never interchangeable. -test("unknown Responses targets reject opaque reasoning and ignore display summaries", () => { +test("unknown Responses targets drop opaque reasoning and preserve display summaries (#10959)", () => { const body: Record = { input: [ { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, @@ -24,11 +24,14 @@ test("unknown Responses targets reject opaque reasoning and ignore display summa ], }; - const originalInput = structuredClone(body.input); const result = applyReasoningInputPolicy(body, "responses"); - assert.equal(result.incompatibleReasoning, true); - assert.deepEqual(body.input, originalInput, "rejection must not mutate the request"); + assert.equal(result.incompatibleReasoning, false); + assert.deepEqual(body.input, [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "reasoning", summary: [{ text: "display only" }] }, + { type: "function_call", name: "search", arguments: "{}", call_id: "call_1" }, + ]); }); test("unannotated targets preserve plaintext Responses reasoning without synthetic IDs", () => { @@ -132,7 +135,7 @@ test("Chat drop removes opaque state while preserving plaintext and summary deta ]); }); -test("DeepSeek rejects plaintext reasoning carrying opaque provider state", () => { +test("DeepSeek projects plaintext reasoning carrying opaque provider state onto the plaintext transport (#10949)", () => { for (const opaqueField of ["signature", "format"] as const) { const body: Record = { input: [ @@ -148,13 +151,11 @@ test("DeepSeek rejects plaintext reasoning carrying opaque provider state", () = const result = applyReasoningInputPolicy(body, "responses", { provider: "deepseek" }); - assert.equal(result.incompatibleReasoning, true, opaqueField); + assert.equal(result.incompatibleReasoning, false, opaqueField); assert.deepEqual(body.input, [ { - id: "rs_mixed123", type: "reasoning", content: [{ type: "reasoning_text", text: "untrusted companion" }], - [opaqueField]: "provider-state", }, { type: "message", role: "user", content: [{ type: "input_text", text: "continue" }] }, ]); @@ -199,6 +200,49 @@ test("drop fallback removes only the incompatible active transport and preserves assert.equal(opaqueReasoning.encrypted_content, "provider-state"); }); +test("mixed plaintext + opaque reasoning follows the target transport instead of rejecting (#10949)", () => { + const mixedReasoning = { + id: "rs_mixed", + type: "reasoning", + content: [{ type: "reasoning_text", text: "inspect first" }], + encrypted_content: "provider-state", + summary: [{ type: "summary_text", text: "display only" }], + }; + + // Plaintext target (deepseek): keep the portable plaintext, strip opaque state. + const toPlaintext: Record = { + input: [structuredClone(mixedReasoning)], + }; + const plaintextResult = applyReasoningInputPolicy(toPlaintext, "responses", { + provider: "deepseek", + }); + assert.equal(plaintextResult.incompatibleReasoning, false); + assert.deepEqual(toPlaintext.input, [ + { + type: "reasoning", + content: [{ type: "reasoning_text", text: "inspect first" }], + summary: [{ type: "summary_text", text: "display only" }], + }, + ]); + + // Opaque target (openai): keep the provider state, strip the plaintext. + const toOpaque: Record = { + input: [structuredClone(mixedReasoning)], + }; + const opaqueResult = applyReasoningInputPolicy(toOpaque, "responses", { + provider: "openai", + }); + assert.equal(opaqueResult.incompatibleReasoning, false); + assert.deepEqual(toOpaque.input, [ + { + id: "rs_mixed", + type: "reasoning", + encrypted_content: "provider-state", + summary: [{ type: "summary_text", text: "display only" }], + }, + ]); +}); + test("drop fallback preserves reasoning when its transport is compatible", () => { const body: Record = { input: [ diff --git a/tests/unit/translator-openai-responses-req.test.ts b/tests/unit/translator-openai-responses-req.test.ts index da87b26116f..2a05f57310e 100644 --- a/tests/unit/translator-openai-responses-req.test.ts +++ b/tests/unit/translator-openai-responses-req.test.ts @@ -172,28 +172,26 @@ test("Responses -> Chat keeps summary-only reasoning out of continuation state", assert.equal(result.messages[0].reasoning_content, undefined); }); -test("Responses -> Chat rejects opaque reasoning instead of replaying its plaintext companion", () => { - assert.throws( - () => - openaiResponsesToOpenAIRequest( - "deepseek-v4-pro", +test("Responses -> Chat replays the plaintext companion of an opaque reasoning item (#10949)", () => { + const result = openaiResponsesToOpenAIRequest( + "deepseek-v4-pro", + { + input: [ { - input: [ - { - id: "rs_opaque", - type: "reasoning", - encrypted_content: "opaque-provider-state", - content: [{ type: "reasoning_text", text: "Untrusted plaintext companion" }], - summary: [{ type: "summary_text", text: "Display summary" }], - }, - { type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, - ], + id: "rs_opaque", + type: "reasoning", + encrypted_content: "opaque-provider-state", + content: [{ type: "reasoning_text", text: "Untrusted plaintext companion" }], + summary: [{ type: "summary_text", text: "Display summary" }], }, - false, - { _preserveReasoningContent: true } - ), - /Reasoning continuation is not compatible/ - ); + { type: "function_call", call_id: "call_1", name: "search", arguments: "{}" }, + ], + }, + false, + { _preserveReasoningContent: true } + ) as { messages: Array> }; + + assert.equal(result.messages[0].reasoning_content, "Untrusted plaintext companion"); }); test("Responses -> Chat merges assistant text that follows a function call", () => { diff --git a/tests/unit/translator-resp-openai-responses.test.ts b/tests/unit/translator-resp-openai-responses.test.ts index 27cc877f2ef..2253c655ac7 100644 --- a/tests/unit/translator-resp-openai-responses.test.ts +++ b/tests/unit/translator-resp-openai-responses.test.ts @@ -423,6 +423,52 @@ test("Responses -> OpenAI: preserves non-object Read JSON-string arguments", () assert.equal(done.choices[0].delta.tool_calls[0].function.arguments, "null"); }); +test("Responses -> OpenAI: mixed plaintext + encrypted_content reasoning replays its plaintext (#10949)", () => { + const state = {}; + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { + type: "reasoning", + id: "rs_mixed", + content: [ + { + type: "reasoning_text", + text: "Let me start by reading the directory to understand the structure of the corpus.", + }, + ], + encrypted_content: "", + summary: [], + }, + }, + state + ); + + assert.ok(done, "mixed reasoning item must surface a delta"); + assert.equal( + done.choices[0].delta.reasoning_content, + "Let me start by reading the directory to understand the structure of the corpus." + ); +}); + +test("Responses -> OpenAI: opaque-only reasoning still emits no fabricated plaintext", () => { + const state = {}; + const done = openaiResponsesToOpenAIResponse( + { + type: "response.output_item.done", + item: { + type: "reasoning", + id: "rs_opaque_only", + encrypted_content: "", + summary: [], + }, + }, + state + ); + + assert.equal(done, null); +}); + test("Responses -> OpenAI: strips empty optional args from JSON-string output_item.done arguments", () => { const state = {}; openaiResponsesToOpenAIResponse( From 2dd20331a7d8d2649d34988c07e22738a0762d2d Mon Sep 17 00:00:00 2001 From: Webman Date: Sat, 22 Aug 2026 21:01:04 -0500 Subject: [PATCH 17/29] feat(providers): add Logfare as a free OpenAI-compatible provider (#10644) (#10987) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged after count reconciliation: the branch's regenerated docs claimed 57 free forever / 157 migrations from its older base; gate-verified values on the current tip are 56 free forever (Logfare carries a Free badge via gateways.ts freeNote but has no freeModelCatalog per-model entries, so the live-code counter stays at 56) and 159 migrations — the README/SVGs now match the check:docs-counts output exactly. provider-consistency OK at 267 registry / 349 canonical. logfare-registry 4/4, icon + KNOWN_PNGS + discovery-set membership all verified present. Closes #10644. Thank you @jonlwheat2-gif! --- AGENTS.md | 2 +- PROVIDER_REFERENCE.md | 447 ++++++++++++++++++ README.md | 25 +- .../features/10987-logfare-free-provider.md | 1 + docs/diagrams/cli-terminal.svg | 2 +- docs/diagrams/comparison-table.svg | 2 +- docs/diagrams/promise-pillars.svg | 6 +- docs/diagrams/readme-hero.svg | 4 +- docs/getting-started/FREE-TIERS-GUIDE.md | 1 + docs/i18n/ar/llm.txt | 8 +- docs/i18n/az/llm.txt | 8 +- docs/i18n/bg/llm.txt | 8 +- docs/i18n/bn/llm.txt | 8 +- docs/i18n/cs/llm.txt | 8 +- docs/i18n/da/llm.txt | 8 +- docs/i18n/de/llm.txt | 8 +- docs/i18n/es/llm.txt | 8 +- docs/i18n/fa/llm.txt | 8 +- docs/i18n/fi/llm.txt | 8 +- docs/i18n/fr/llm.txt | 8 +- docs/i18n/gu/llm.txt | 8 +- docs/i18n/he/llm.txt | 8 +- docs/i18n/hi/llm.txt | 8 +- docs/i18n/hu/llm.txt | 8 +- docs/i18n/id/llm.txt | 8 +- docs/i18n/in/llm.txt | 8 +- docs/i18n/it/llm.txt | 8 +- docs/i18n/ja/llm.txt | 8 +- docs/i18n/ko/llm.txt | 8 +- docs/i18n/mr/llm.txt | 8 +- docs/i18n/ms/llm.txt | 8 +- docs/i18n/nl/llm.txt | 8 +- docs/i18n/no/llm.txt | 8 +- docs/i18n/phi/llm.txt | 8 +- docs/i18n/pl/llm.txt | 8 +- docs/i18n/pt-BR/llm.txt | 8 +- docs/i18n/pt/llm.txt | 8 +- docs/i18n/ro/llm.txt | 8 +- docs/i18n/ru/llm.txt | 8 +- docs/i18n/sk/llm.txt | 8 +- docs/i18n/sv/llm.txt | 8 +- docs/i18n/sw/llm.txt | 8 +- docs/i18n/ta/llm.txt | 8 +- docs/i18n/te/llm.txt | 8 +- docs/i18n/th/llm.txt | 8 +- docs/i18n/tr/llm.txt | 8 +- docs/i18n/uk-UA/llm.txt | 8 +- docs/i18n/ur/llm.txt | 8 +- docs/i18n/vi/llm.txt | 8 +- docs/i18n/zh-CN/llm.txt | 8 +- docs/i18n/zh-TW/llm.txt | 8 +- docs/reference/PROVIDER_REFERENCE.md | 21 +- llm.txt | 8 +- open-sse/config/providers/index.ts | 2 + .../providers/registry/logfare/index.ts | 25 + package.json | 2 +- promise-pillars.svg | 139 ++++++ public/providers/logfare.png | Bin 0 -> 17858 bytes .../[id]/models/discovery/providerSets.ts | 5 + src/shared/components/ProviderIcon.tsx | 1 + src/shared/constants/providers.ts | 1 + .../constants/providers/apikey/gateways.ts | 23 + tests/snapshots/provider/translate-path.json | 23 + tests/unit/logfare-registry.test.ts | 63 +++ tests/unit/providers-constants-split.test.ts | 12 +- 65 files changed, 942 insertions(+), 209 deletions(-) create mode 100644 PROVIDER_REFERENCE.md create mode 100644 changelog.d/features/10987-logfare-free-provider.md create mode 100644 open-sse/config/providers/registry/logfare/index.ts create mode 100644 promise-pillars.svg create mode 100644 public/providers/logfare.png create mode 100644 tests/unit/logfare-registry.test.ts diff --git a/AGENTS.md b/AGENTS.md index f10fb39483c..d468f7d1984 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -46,7 +46,7 @@ Repository map and Reference Documentation sections below. ## Project at a Glance -**OmniRoute** — unified AI proxy/router. One endpoint, 348 LLM providers, auto-fallback. +**OmniRoute** — unified AI proxy/router. One endpoint, 349 LLM providers, auto-fallback. | Layer | Location | Purpose | | ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | diff --git a/PROVIDER_REFERENCE.md b/PROVIDER_REFERENCE.md new file mode 100644 index 00000000000..571fe0e904c --- /dev/null +++ b/PROVIDER_REFERENCE.md @@ -0,0 +1,447 @@ +--- +title: "Provider Reference" +version: 3.8.50 +lastUpdated: 2026-08-21 +--- + +# Provider Reference + +> **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. +> Regenerate with: `npm run gen:provider-reference` +> **Last generated:** 2026-08-21 + +Total providers: **349**. See category breakdown below. + +## Categories + +- **Free** — free tier with API key (configured via dashboard) +- **No-auth** — public endpoints that require no key or sign-in at all +- **OAuth** — sign-in flow handled by OmniRoute, no API key needed +- **Web cookie** — wraps the provider's web app via cookie auth +- **API key** — paid provider configured via API key (free credits may apply) +- **Local** — runs on the user's machine (Ollama, LM Studio, vLLM, etc.) +- **Search** — web search providers +- **Audio** — audio-only providers (TTS/STT) +- **Upstream proxy** — providers that proxy to other providers +- **Cloud agent** — long-running coding agents (Codex Cloud, Devin, Jules) +- **System** — OmniRoute-internal providers (loopback, etc.) + +Additional tags: `image`, `video`, `aggregator`, `enterprise`, `embed/rerank`, `self-hosted`. + +`Tool calling` (where shown): `native` — real function-calling API; `emulated` — the `tools` array is prompt-emulated via `webTools.ts` (regex-parsed `{...}` blocks); `none` — `tools` is currently silently dropped. See #7286. + +Use the dashboard at `/dashboard/providers` to enable, configure, and test each provider. + +--- + +## No-auth Providers (no key required) (11) + +| ID | Alias | Name | Tags | Website | Notes | Tool calling | +|----|-------|------|------|---------|-------|--------------| +| `aihorde` | `horde` | AI Horde | No-auth | [link](https://aihorde.net) | No API key required — uses AI Horde's documented anonymous key. Adding a free aihorde.net key is optional and only buys higher queue priority (kudos). | — | +| `auggie` | `aug` | Augment (Auggie CLI) | No-auth | [link](https://augmentcode.com) | No API key stored by OmniRoute. Install the Auggie CLI and run `auggie login` on this machine, then OmniRoute spawns it locally for each request. | — | +| `chipotle` | `pepper` | Chipotle Pepper AI (Free) | No-auth | [link](https://amelia.chipotle.com) | No credentials required. Uses Chipotle's public support chatbot via reverse-engineered SockJS/STOMP protocol. | — | +| `cloudflare-playground` | `cfp` | Cloudflare AI Playground | No-auth | [link](https://playground.ai.cloudflare.com) | No credentials required — anonymous browser sessions over a reverse-engineered cf_agent WebSocket protocol (Playwright transport). | — | +| `devin-cli-agentic` | `dva` | Devin CLI Agentic Bridge | No-auth | [link](https://docs.devin.ai/work-with-devin/devin-cli) | Authentication is owned by the official Devin CLI in its isolated bridge volume. | emulated | +| `duckduckgo-web` | `ddgw` | DuckDuckGo AI Chat | No-auth | [link](https://duckduckgo.com/duckchat) | No credentials required — DuckDuckGo AI Chat is anonymous and free. | emulated | +| `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — | +| `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — | +| `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — | +| `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — | +| `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — | + +## OAuth Providers (25) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | +| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | +| `antigravity` | — | Antigravity | OAuth | — | — | +| `claude` | `cc` | Claude Code | OAuth | — | — | +| `cline` | `cl` | Cline | OAuth | — | — | +| `clinepass` | `cp` | ClinePass | OAuth | [link](https://cline.bot/cline-pass) | ClinePass is Cline's $9.99/mo subscription bundling 10 open coding models. Sign in with your Cline account (same login as the Cline CLI/IDE), or paste a direct ClinePass API key (app.cline.bot → Settings → API Keys). A ClinePass subscription unlocks the cline-pass/* models. Reuses the Cline WorkOS OAuth flow. | +| `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. | +| `codex` | `cx` | OpenAI Codex | OAuth | — | — | +| `cursor` | `cu` | Cursor IDE | OAuth | — | — | +| `devin-cli` | `dv` | Devin CLI | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | +| `devin-desktop` | — | Devin Desktop | OAuth | [link](https://devin.ai) | Paste an existing Devin API key from an authenticated Devin session. Key export availability and steps vary by Devin version and account. | +| `ghe-copilot` | `ghe-copilot` | GitHub Enterprise Copilot | OAuth | — | Enter your GHE instance URL (e.g., https://ghe.company.com) in provider settings, then authenticate via device flow. | +| `github` | `gh` | GitHub Copilot | OAuth | — | — | +| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab Duo OAuth is not configured. Register an OAuth application at https://gitlab.com/-/profile/applications with redirect URI http://localhost:20128/callback and scopes "ai_features read_user", then set GITLAB_DUO_OAUTH_CLIENT_ID (and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET) and restart. | +| `grok-cli` | `gc` | Grok Build | OAuth | — | Sign in with your browser, or paste your ~/.grok/auth.json (or the JWT access token) from the Grok Build CLI; refresh_token is rotated automatically either way. | +| `kilocode` | `kc` | Kilo Code | OAuth | — | — | +| `kimi-coding` | `kmc` | Kimi Code CLI | OAuth | [link](https://www.kimi.com/code?aff=omniroute) | Sign in with the same Kimi account used by Kimi Code CLI. OmniRoute uses the CLI OAuth flow and Kimi Coding Plan endpoints. | +| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | +| `openference` | `of` | Openference | OAuth | [link](https://openference.com) | Sign in with your Openference account to route requests through api.openference.com. An active plan is required for inference — OAuth may authenticate but return 402 without one. | +| `qoder` | `if` | Qoder | OAuth | — | — | +| `raycast` | `rc` | Raycast Pro AI | OAuth | [link](https://raycast.com/ai) | Unofficial integration — uses your Raycast Pro subscription via credentials from the macOS app (Auto-Import or manual capture). May break on Raycast updates. Not for redistribution; personal use only. | +| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | +| `xai-oauth` | `xao` | xAI OAuth (Grok) | OAuth | [link](https://x.ai) | Sign in with xAI to use api.x.ai models such as Grok 4.5. This is separate from Grok Build JWT sessions, which use cli-chat-proxy.grok.com and grok-build model aliases. | +| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | +| `zed-hosted` | — | Zed Hosted Models | OAuth | [link](https://zed.dev) | Sign in with your Zed account (native-app sign-in). OmniRoute generates a one-time RSA keypair and opens zed.dev to authorize it — on a remote/headless install, copy the resulting 127.0.0.1 callback URL from your browser's address bar and paste it back here. Distinct from the 'Zed IDE' credential-import entry above: this proxies chat completions through Zed's own hosted model aggregator (cloud.zed.dev), fronting Anthropic/OpenAI/Google/xAI models under your Zed plan. | + +## Web Cookie Providers (35) + +| ID | Alias | Name | Tags | Website | Notes | Tool calling | +|----|-------|------|------|---------|-------|--------------| +| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | emulated | +| `adobe-firefly` | `firefly` | Adobe Firefly (Image/Video) | Web cookie | [link](https://firefly.adobe.com) | RECOMMENDED: firefly.adobe.com signed-in → F12 → Network → click firefly-3p.ff.adobe.io (generate-async or models/discovery) → Request Headers → Authorization → copy the token AFTER 'Bearer ' (starts with eyJ…). Cookie-only from firefly.adobe.com mints a GUEST token → 401/403; only multi-domain IMS cookies (adobelogin.com) or that Bearer JWT work. Unofficial/experimental media + Limits. | — | +| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | emulated | +| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | emulated | +| `chatgpt-web-codex` | `cgpt-codex` | ChatGPT Web (Codex) | Web cookie | [link](https://chatgpt.com) | Paste the full ChatGPT Cookie header. OmniRoute verifies it in an isolated headless browser profile. | native | +| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | none | +| `conol-web` | `cnl` | Conol (Unofficial/Experimental) | Web cookie | [link](https://conol.ai) | Use browser sign-in, or paste the full Cookie header from conol.ai. The __Secure-better-auth.session_token cookie is required. | — | +| `copilot-m365-web` | `m365copilot` | Microsoft 365 Copilot (BizChat) | Web cookie | [link](https://m365.cloud.microsoft/chat) | Sign in at m365.cloud.microsoft/chat, then open DevTools → Network → filter 'WS' → click the Chathub WebSocket connection. Copy both the access_token query parameter AND the account-specific Chathub path segment from its request URL (wss://…/Chathub/?…&access_token=…). It is NOT an Authorization: Bearer header on an XHR/Fetch request. The token is short-lived; this is an unofficial integration. Optional: store a refresh_token in providerSpecificData.refreshToken (any Microsoft device-code/refresh flow for the substrate.office.com/sydney scopes) and OmniRoute pre-flight-refreshes the access token itself — otherwise re-capture after every ~75 min expiry. | — | +| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste the access_token from an authenticated copilot.microsoft.com request (DevTools → Network → Authorization), or export a HAR while logged in | — | +| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | emulated | +| `doubao-web` | `db` | Dola Web (ByteDance) | Web cookie | [link](https://www.dola.com) | Paste the full Cookie header from www.dola.com. It should include sessionid, ttwid, and s_v_web_id. If s_v_web_id is unavailable, fp=verify_... from a chat/completion request URL can be used as a fallback. | — | +| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — | +| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated | +| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — | +| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | +| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — | +| `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — | +| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated | +| `kimi-web` | `kimi-web` | Kimi Web | Web cookie | [link](https://www.kimi.com/code?aff=omniroute) | Paste access_token from www.kimi.com DevTools → Application → Local Storage. A legacy kimi-auth cookie is also accepted. | — | +| `lmarena` | `lma` | Arena (Free) | Web cookie | [link](https://arena.ai) | Paste the full Cookie header from arena.ai (DevTools → Network → request → Cookie). Include arena-auth-prod-v1.0/.1… and cf_clearance/__cf_bm when present. OmniRoute uses Chrome TLS impersonation; if Arena still 403s, set providerSpecificData.recaptchaV3Token from a live browser session. | — | +| `microsoft-designer-web` | `msdesigner` | Microsoft Designer (Image Generation) | Web cookie | [link](https://designer.microsoft.com) | Sign in at designer.microsoft.com, then open DevTools → Network, generate an image, and find the request to DallE.ashx?action=GetDallEImagesCogSci. Copy the value of its Authorization: Bearer header (the access_token — no 'Bearer ' prefix). The token is short-lived; this is an unofficial, reverse-engineered integration. | — | +| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your ecto_1_sess cookie AND the ecto1:... WS auth token from meta.ai. Capture the ecto1: token in DevTools → Network → WS → the clippy request's Authorization query param. Example: ecto_1_sess=4240a308...NVDg0; ecto1:ABCD... | emulated | +| `notion-web` | `nw` | Notion AI Web (Unofficial/Experimental) | Web cookie | [link](https://www.notion.so) | Paste only the token_v2 cookie VALUE from app.notion.com (DevTools → Application → Cookies → token_v2). Do not paste token_v2= or the full Cookie header. Workspace is auto-detected; space_id / notion_user_id are optional. | — | +| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | emulated | +| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | — | +| `promptql` | `pql` | PromptQL (Unofficial/Experimental) | Web cookie | [link](https://prompt.ql.app) | Paste the Bearer JWT from prompt.ql.app DevTools → Network → graphql → Authorization (token only). Optional projectId + session Cookie for refresh. | — | +| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | emulated | +| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | emulated | +| `tencent-aistudio-web` | `tasw` | Tencent AI Studio (Free) | Web cookie | [link](https://aistudio.tencent.ai) | Log in to aistudio.tencent.ai, open DevTools -> Network, copy any request Cookie header containing session tokens. | — | +| `tinycms-web` | `tcw` | TinyCMS Web (Free/Sub) | Web cookie | [link](https://site.tinycms.xyz) | Go to site.tinycms.xyz, open DevTools → Application → Local Storage, copy the value of 'app-config-uuid' (starts with 'R'), and paste it here. | — | +| `v0-vercel-web` | `v0-vercel-web` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | — | +| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | — | +| `yuanbao-web` | `ybw` | Tencent Yuanbao (Free) | Web cookie | [link](https://yuanbao.tencent.com) | Log in to yuanbao.tencent.com, then paste the full Cookie header (DevTools → Network → any /api request → Request Headers → Cookie). It must contain hy_user and hy_token. | — | +| `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | +| `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | + +## API Key Providers (paid / paid-with-free-credits) (233) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | +| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | +| `agnes` | `agnes` | Agnes AI | API key, video | [link](https://agnes-ai.com) | Get API key at agnes-ai.com | +| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | +| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | Free tier paused (2026) — AI/ML API is now pay-as-you-go only (min $20 top-up); no recurring free credits. | +| `ainative` | `ainative` | AINative Studio | API key | [link](https://ainative.studio) | Create a free API key at ainative.studio (no card), then paste it here as a Bearer token. | +| `aion` | `aion` | Aion Labs | API key | [link](https://www.aionlabs.ai) | Create a free API key at aionlabs.ai (no card), then paste it here as a Bearer token. | +| `alibaba` | `ali` | Alibaba Cloud Model Studio | API key | [link](https://bailian.console.alibabacloud.com/) | — | +| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.console.aliyun.com/) | — | +| `ant-ling` | `ling` | Ant Ling / Ring (inclusionAI) | API key | [link](https://developer.ant-ling.com/en/docs/) | Register and create an API key at the Ant Ling API console (https://chat.ant-ling.com/open), then paste it here. OmniRoute routes chat traffic to https://api.ant-ling.com/v1/chat/completions; the provider is OpenAI-compatible and also exposes an Anthropic-compatible surface. | +| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | +| `anyapi` | `anyapi` | AnyAPI AI | API key, aggregator | [link](https://anyapi.ai) | Free plan: 100,000 ANY Tokens/day and 100 RPM for eligible Free/Basic models; no credit card required. | +| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | +| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | +| `auriko` | `auriko` | Auriko | API key, aggregator | [link](https://www.auriko.ai) | Free plan publishes 1,000 Platform RPM and 10,000 BYOK RPM. Platform inference still passes through provider cost; this is not a free-token pool or unlimited free inference. | +| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | +| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | +| `bai` | `bai` | b.ai | API key | [link](https://b.ai) | Bearer API key for the b.ai OpenAI-compatible LLM gateway (distinct from TheB.AI). Create a key at https://docs.b.ai, then use https://api.b.ai/v1 as the OpenAI-compatible base URL. | +| `baichuan` | `baichuan` | Baichuan | API key | [link](https://www.baichuan-ai.com/) | Get API key at platform.baichuan-ai.com | +| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://ernie.baidu.com/) | Get API key at console.bce.baidu.com | +| `bailian-coding-plan` | `bcp` | Alibaba Token Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/token-plan-overview) | — | +| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | +| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | +| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | +| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply | +| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | +| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | +| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | +| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | +| `charm-hyper` | `charm-hyper` | Charm Hyper | API key | [link](https://hyper.charm.land) | 100 free monthly Hypercredits on signup | +| `chat-oripe` | `chat-oripe` | Chat Oripe | API key, aggregator | [link](https://api.oriper.com) | Official metadata advertises 2M tokens/month, but the public site and documentation were blocked during audit; treat the quota and brand mapping as unconfirmed. | +| `chatanywhere` | `chatanywhere` | ChatAnywhere | API key, aggregator | [link](https://chatanywhere.tech) | Personal, educational or research use only: public documentation cites 10,000 points/day and 200 requests/day per IP/key; do not use for commercial traffic. | +| `cheaperinference` | `cinf` | Cheaper Inference | API key | [link](https://cheaperinference.com/?utm_source=omniroute) | — | +| `chenzk` | `chenzk` | Chenzk API | API key | [link](https://chenzk.top) | — | +| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | +| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | +| `cloudcode-one` | `cloudcode-one` | CloudCode.ONE | API key, aggregator | [link](https://cloudcode.one) | Published free models include glm-4.7-flash and glm-4.6v-flash; no numeric quota is published, and key creation may require credit or a coupon. | +| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | +| `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — | +| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | +| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | +| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | +| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | +| `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. | +| `dahl` | `dahl` | Dahl | API key | [link](https://inference.dahl.global) | Click 'Add Account' to auto-generate a token, or add a manual API key. | +| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | +| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | +| `deepai` | `deepai` | DeepAI | API key, image | [link](https://deepai.org) | Use your DeepAI API key. Get one at deepai.org — requires a Pro subscription ($9.99/mo). | +| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | +| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | +| `dgrid` | `dgrid` | DGrid | API key | [link](https://dgrid.ai) | DGrid Free Models Router: 10 requests/minute and 100 requests/day. A $5 lifetime top-up unlocks up to 20 requests/minute and 1,000 requests/day. | +| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | +| `digitalocean` | `digitalocean` | DigitalOcean | API key | [link](https://docs.digitalocean.com/products/ai-platform/) | — | +| `dit` | `dai` | DIT.ai | API key | [link](https://dit.ai) | Use your dit.ai API key in Authorization: Bearer . Fully OpenAI-compatible — a drop-in replacement, just change the base URL to https://api.dit.ai/v1. | +| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | +| `dxnt` | `dxnt` | DXNT / DX Token | API key, aggregator | [link](https://www.dxnt.com) | Free accounts are documented at 100 calls/day; the quota may increase through invitations and can vary by account. | +| `electronhub` | `electronhub` | Electron Hub | API key, aggregator | [link](https://www.electronhub.ai) | Free plan: 5 RPM, $0.25 weekly credits and 10 Neutrinos/day for :free models; family budgets also apply. | +| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | +| `factory` | `factory` | Factory | API key | [link](https://factory.ai) | Bearer API key for the Factory OpenAI-compatible gateway. | +| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | +| `fastrouter` | `fastrouter` | FastRouter | API key, aggregator | [link](https://fastrouter.ai) | Models with the :free suffix allow 10 requests/day per organization and model; availability may change. | +| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | +| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | +| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | +| `free-ai` | `free-ai` | Free.ai | API key, aggregator | [link](https://free.ai) | 30,000 tokens/day cover self-hosted models after email verification. Usage beyond the pool can bill at raw cost, and premium external models are paid. | +| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | +| `freebuff` | `freebuff` | Freebuff | API key | [link](https://freebuff.com) | Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester). | +| `freeinference` | `freeinference` | FreeInference | API key, aggregator | [link](https://freeinference.org) | Free research access without a card; non-Harvard applicants require manual approval and no numeric quota is publicly guaranteed. | +| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | +| `freetheai` | `fta` | FreeTheAi | API key, aggregator | [link](https://freetheai.xyz) | Join the FreeTheAi Discord to get your free API key. | +| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | +| `g4f-gemini` | `g4fgem` | g4f.space — Gemini | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-groq` | `g4fgroq` | g4f.space — Groq | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-nvidia` | `g4fnv` | g4f.space — NVIDIA | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-ollama` | `g4foll` | g4f.space — Ollama | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `g4f-pollinations` | `g4fpol` | g4f.space — Pollinations | API key, aggregator | [link](https://g4f.space) | No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits. | +| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | ⚠️ **DEPRECATED.** api.galadriel.ai no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | +| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free tier available through Google AI Studio; current per-model quotas and regional limits apply | +| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | +| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | +| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | +| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free MiMo (xiaomi/mimo-v2.5) revoked 2026-05 — Opengateway is now a pay-as-you-go credit gateway; no recurring free model. | +| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free Nemotron promo ended 2026-06 — the GMI Cloud route is now pay-as-you-go credit only. | +| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | +| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | +| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | +| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | +| `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | +| `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | +| `helyxai` | `helyxai` | Helyx AI | API key, aggregator | [link](https://helyxai.space) | Operational Free plan documents 100,000 tokens/day; the site's separate 2M+ marketing claim conflicts and is not treated as a quota guarantee. | +| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | +| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | +| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | +| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | +| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `inception` | `inception` | Inception | API key | [link](https://docs.inceptionlabs.ai) | 10M free tokens on signup, no credit card required. | +| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | +| `internlm` | `internlm` | InternLM (Intern-S1) | API key | [link](https://internlm.intern-ai.org.cn/) | Free monthly quota ~1M input / 3M output tokens (~10 RPM) | +| `jina-ai` | `jina` | Jina AI (Foundation API) | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for api.jina.ai — embeddings, rerank, classify, segment, and search. Dashboard keys take precedence over JINA_AI_API_KEY. This is not the Reader / r.jina.ai card and does not fetch URLs. | +| `jina-reader` | `jr` | Jina Reader (r.jina.ai) | API key | [link](https://jina.ai/reader) | Bearer API key for r.jina.ai URL-to-markdown (/v1/web/fetch only). Does not serve /v1/embeddings or /v1/rerank. The same Jina token as Foundation API works; OmniRoute reuses a jina-ai dashboard key or JINA_AI_API_KEY when this card is empty. | +| `kenari` | `kenari` | Kenari | API key | [link](https://kenari.id) | Use your Kenari API key (kn-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://kenari.id/v1. | +| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | +| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | +| `kimi` | `kimi` | Kimi (Legacy Moonshot API) | API key | [link](https://platform.kimi.ai?aff=omniroute) | — | +| `kimi-coding-apikey` | `kmca` | Kimi Code API Key | API key | [link](https://www.kimi.com/code?aff=omniroute) | — | +| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | +| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | +| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | +| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | +| `literouter` | `literouter` | LiteRouter | API key, aggregator | [link](https://literouter.com) | Free model variants use the :free suffix; daily credit limits vary by model and free input is capped at 5,000 tokens. | +| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | +| `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. | +| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | +| `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | +| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. | +| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | +| `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | +| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | +| `meganova-ai` | `meganova-ai` | MegaNova AI | API key, aggregator | [link](https://meganova.ai) | Free signup without a card. Published Tier 1 per-model quotas total 550 requests/day; they are not a shared global pool, and paid overage can apply if enabled. | +| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | +| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | +| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | +| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | +| `mixedbread` | `mxbai` | Mixedbread AI | API key | [link](https://www.mixedbread.com) | Bearer API key for the Mixedbread embeddings API. | +| `mixlayer` | `mixlayer` | Mixlayer | API key, aggregator | [link](https://www.mixlayer.com) | The qwen/qwen3.5-4b-free model is free for prototyping and rate-limited; no fixed public RPM or daily quota is confirmed. | +| `mnn-ai` | `mnn-ai` | MNN AI | API key, aggregator | [link](https://mnnai.ru) | Free plan: $1 monthly credits, 10 RPM and access only to models marked Free. | +| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | +| `modelscope` | `ms` | ModelScope | API key | [link](https://modelscope.cn) | Free tier via ModelScope API-Inference — Alibaba account required. | +| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | ⚠️ **DEPRECATED.** Monster API shuttered operations on 2026-06-30. Use alternative OpenAI-compatible providers. | +| `moonshot` | `moonshot` | Kimi | API key | [link](https://platform.kimi.ai?aff=omniroute) | — | +| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | +| `muse-code` | `mc` | Muse Code (Meta) | API key | [link](https://github.com/meta-llama/llama-stack) | Use your META_API_KEY env var as a Bearer token. Muse Code CLI uses the OpenAI Responses API wire format (POST /responses). | +| `naga-ac` | `naga` | Naga.ac | API key, aggregator | [link](https://naga.ac) | Get API key at naga.ac — Google/GitHub/Discord signup available. | +| `naga-ai` | `naga-ai` | Naga AI | API key, aggregator | [link](https://naga.ac) | Models marked :free are publicly listed, but no numeric quota is confirmed. Naga's policy warns that free-tier prompts and outputs may be collected or used for training. | +| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | +| `nara` | `nara` | NaraRouter | API key | [link](https://bynara.id) | Get a free API key via NaraRouter's Telegram channel, then paste it here as a Bearer token. | +| `navy` | `navy` | NavyAI | API key | [link](https://api.navy) | Create a free API key from the NavyAI dashboard, then paste it here as a Bearer token. | +| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | +| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | +| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | +| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | +| `novita` | `novita` | Novita AI | API key, video, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | +| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | +| `nube` | `nube` | Nube.sh | API key | [link](https://nube.sh) | — | +| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | +| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | +| `ofoxai` | `ofoxai` | OfoxAI | API key, aggregator | [link](https://ofox.ai) | The current catalog advertises 10+ free models without a public numeric quota; review upstream provenance, retention and training terms before production use. | +| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/keys) | — | +| `openadapter` | `oad` | OpenAdapter | API key | [link](https://openadapter.dev) | Use your OpenAdapter API key in Authorization: Bearer sk-cv-. Fully OpenAI-compatible. API base URL: https://api.openadapter.in/v1. | +| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | +| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | +| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | +| `openference-api` | `ofa` | Openference API | API key | [link](https://openference.com) | Free plan: 3-day trial with open-source models — no credit card required | +| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | +| `openvecta` | `openvecta` | OpenVecta | API key | [link](https://openvecta.com) | Free credits on signup for OpenAI-compatible inference across LLMs, embeddings, and reasoning models | +| `orcarouter` | `orcarouter` | OrcaRouter | API key | [link](https://www.orcarouter.ai) | — | +| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | +| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | +| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | +| `pioneer` | `pn` | Pioneer AI | API key | [link](https://pioneer.ai) | $75 free usage credits — no credit card required | +| `plamo` | `plamo` | PLaMo | API key | [link](https://plamo.preferredai.jp/api) | — | +| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | +| `poixe-ai` | `poixe-ai` | Poixe AI | API key, aggregator | [link](https://poixe.com) | Current public free limits are small and model-group specific: 2 RPM/5 RPD for large-cup models and 20 RPM/50 RPD for small-cup models. | +| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | Anonymous/keyless access to the documented free models is best-effort. Local v3.8.50 verification (2026-07-31) returned 401 via OmniRoute and Cloudflare 1010 on direct upstream probes from the same network. Premium models still require a Pollinations API key from enter.pollinations.ai. | +| `poolside` | `poolside` | Poolside | API key | [link](https://poolside.ai) | Laguna S 2.1 and XS 2.1 are free during Preview; no public numeric quota is published. | +| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | ⚠️ **DEPRECATED.** serving.app.predibase.com no longer resolves (sweep 2026-06-19); the managed serving API appears discontinued. | +| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid | +| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product-s/qianfan_home) | — | +| `qiniu` | `qiniu` | Qiniu | API key | [link](https://www.qiniu.com) | — | +| `qwen-cloud` | `qwc` | Qwen Cloud | API key | [link](https://www.qwencloud.com/) | — | +| `qwen-cloud-token-plan` | `qct` | Qwen Cloud Token Plan | API key | [link](https://www.qwencloud.com/pricing/token-plan) | — | +| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | +| `regolo` | `regolo` | Regolo AI | API key | [link](https://regolo.ai) | Get your Regolo API key from regolo.ai, then paste it here as a Bearer token. | +| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | +| `requesty` | `requesty` | Requesty | API key | [link](https://requesty.ai) | Free tier ~200 requests/day - multi-model routing gateway (300+ models) | +| `routeway` | `routeway` | Routeway | API key | [link](https://routeway.ai) | Create a free API key at routeway.ai, then paste it here as a Bearer token. | +| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | +| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | +| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | +| `sarvam` | `sarvam` | Sarvam AI | API key | [link](https://docs.sarvam.ai) | ₹1,000 in free signup credits — never expire | +| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/docs/ai-data/generative-apis/) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | +| `sealion` | `sealion` | SEA-LION | API key | [link](https://sea-lion.ai) | Sign in at sea-lion.ai with Google (no card, no region wall), create an API key, then paste it here. | +| `segmind` | `segmind` | Segmind | API key, image, video | [link](https://segmind.com) | Use your Segmind API key in the x-api-key header. OmniRoute targets https://api.segmind.com/v1/ and returns the generated image/video bytes directly. | +| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | +| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus currently listed $0 models after identity verification; availability and limits may change | +| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | +| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `speka` | `speka` | Speka AI | API key, aggregator | [link](https://speka.me) | Free plan: $1 monthly usage, 10 RPM, one API key and access to open models and the playground; no card required. | +| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | +| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | +| `sumopod` | `sumopod` | SumoPod | API key | [link](https://ai.sumopod.com) | Use your SumoPod API key (sk-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://ai.sumopod.com/v1. | +| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | +| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | +| `tabitoken` | `tabitoken` | TabiToken | API key, aggregator | [link](https://tabitoken.com) | — | +| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | +| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | +| `tinyfish` | `tf` | TinyFish Fetch | API key | [link](https://docs.tinyfish.ai/fetch-api) | X-API-Key from agent.tinyfish.ai/api-keys | +| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | — | +| `token-kiosk` | `tk` | Token Kiosk | API key | [link](https://agent-router.gaib.ai) | Use your Token Kiosk API key in Authorization: Bearer . Fully OpenAI-compatible gateway. API base URL: https://agent-router.gaib.ai/v1. | +| `tokenreply` | `tokenreply` | TokenReply | API key, aggregator | [link](https://www.tokenreply.com) | Free-tagged models have model- and campaign-specific daily limits; no fixed global free quota is published. | +| `tokenrouter` | `trk` | TokenRouter | API key | [link](https://tokenrouter.com) | Use your TokenRouter API key in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.tokenrouter.com/v1. | +| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | +| `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. | +| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | +| `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. | +| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | +| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | +| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | +| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | +| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | +| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | +| `void-ai` | `void-ai` | Void AI | API key, aggregator | [link](https://voidai.app) | The public model catalog marks some models with a free plan requirement, but access is conditional and no numeric quota is confirmed. | +| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | +| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | +| `wafer` | `wafer` | Wafer AI | API key | [link](https://wafer.ai) | — | +| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | +| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | +| `writer` | `writer` | Writer | API key | [link](https://dev.writer.com) | — | +| `x5lab` | `x5lab` | X5Lab | API key | [link](https://x5lab.dev) | Use your X5Lab API key (x5-...) in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.x5lab.dev/v1. | +| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | Use an official xAI API key, or sign in with xAI OAuth. Grok Build JWT sessions remain a separate provider. | +| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | +| `xiaomi-mimo-token-plan` | `mimotp` | Xiaomi MiMo Token Plan | API key | [link](https://mimo.mi.com) | — | +| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | +| `yolo-auto` | `yolo-auto` | Yolo-Auto | API key, aggregator | [link](https://yolo-auto.com) | Free API access is request-limited and intended for testing; no numeric daily quota is published and free access is not promised indefinitely. | +| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | +| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. | +| `zerolimitai` | `zerolimitai` | ZeroLimitAI | API key, aggregator | [link](https://www.zerolimitai.com) | Temporary free trial is advertised, but official pages conflict between 3 and 7 days; a 100-calls/day claim is not treated as permanent. | +| `zylo-api` | `zylo` | Zylo API | API key, aggregator | [link](https://zyloai.net) | Basic plan: 10 RPM, 7,200 requests/day and 200,000 tokens/day; limited to Basic text models. | + +## Local Providers (14) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | +| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | +| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | +| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | +| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | +| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | +| `mlx-gemma` | `mlx-gemma` | MLX Gemma 26B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11435. Requires uv and mlx-lm installed. Model: mlx-community/gemma-4-26B-A4B-it-qat-q4_0-mlx-aligned (~15.9GB peak memory). | +| `mlx-qwen` | `mlx-qwen` | MLX Qwen 3.8 27B | Local, self-hosted | [link](https://github.com/ml-explore/mlx) | No API key required. Runs mlx-lm server locally on port 11436. Requires uv and mlx-lm installed. Model: maglun/Qwen3.8-27B-MLX-Mixed-3.80bpw (~13.1GB peak memory). | +| `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. | +| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | +| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | +| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | + +## Search Providers (13) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | +| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | +| `firecrawl` | `fc` | Firecrawl | Search | [link](https://firecrawl.dev) | API key from firecrawl.dev/app/api-keys (or set your self-hosted Firecrawl base URL) | +| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | +| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | +| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/keys) | Same API key as Ollama Cloud (from ollama.com/settings/keys) | +| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | +| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs/google) | API key from SearchAPI (query param or Bearer auth) | +| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | +| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | +| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | +| `x-search` | `x_search` | X Search (Grok) | Search | [link](https://docs.x.ai/developers/tools/x-search) | SuperGrok OAuth (xai-oauth) or xAI API key. This is Grok X Search, not the X Developer MCP. | +| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/business/api/) | X-API-Key from the You.com platform dashboard | + +## Audio-only Providers (12) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | +| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | +| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | +| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | +| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | +| `fishaudio` | `fishaudio` | Fish Audio | Audio | [link](https://fish.audio) | — | +| `gladia` | `gladia` | Gladia | Audio | [link](https://gladia.io) | — | +| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | +| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | +| `rev-ai` | `revai` | Rev AI | Audio | [link](https://www.rev.ai) | — | +| `soniox` | `sx` | Soniox | Audio | [link](https://soniox.com) | — | +| `speechmatics` | `sm` | Speechmatics | Audio | [link](https://www.speechmatics.com) | Free tier — 8 hours/month, no credit card required. Batch (async) mode only. | + +## Upstream Proxy Providers (2) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | +| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | + +## Cloud Agent Providers (3) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | +| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | +| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | + +## System Providers (1) + +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `auto` | `auto` | Auto (Zero-Config) | System | — | — | + +## Sources of truth + +- Catalog: [`src/shared/constants/providers.ts`](../../src/shared/constants/providers.ts) +- Registry (per-model details): [`open-sse/config/providerRegistry.ts`](../../open-sse/config/providerRegistry.ts) +- Executors: [`open-sse/executors/`](../../open-sse/executors/) (106 implementations) +- Translators: [`open-sse/translator/`](../../open-sse/translator/) + +## See Also + +- [FREE_TIERS.md](./FREE_TIERS.md) — curated free-tier guide +- [USER_GUIDE.md](../guides/USER_GUIDE.md) — provider setup walkthrough +- [ARCHITECTURE.md](../architecture/ARCHITECTURE.md) — overall architecture diff --git a/README.md b/README.md index 55e6d0328a1..6600b406492 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 348 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 348 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 349 providers — 90+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 349 AI providers · 90+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start.
@@ -101,7 +101,7 @@ ⚙️ Features 🎯 Combos - 🌐 Providers + 🌐 Providers 🔌 CLI & MCP @@ -210,7 +210,7 @@ curl http://localhost:20128/v1/chat/completions \ -The Promise — One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 348 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests). +The Promise — One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 349 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests).

@@ -461,7 +461,7 @@ All **19** strategies — mix & match per combo step: -What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 348 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. +What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 349 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs. 📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md) @@ -559,7 +559,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute - **🖼️ New endpoints** — `/v1/ocr` (Mistral OCR) and `/v1/audio/translations` (Whisper-style) round out the media surface. → [API Reference](docs/reference/API_REFERENCE.md) - **🎨 Image / video / audio generation** — one API for media: xAI Grok Imagine & Novita AI video, ComfyUI, Freepik, Adobe Firefly, Microsoft Designer, Segmind, EdgeTTS. → [API Reference](docs/reference/API_REFERENCE.md) - **🌍 Deployment & ops** — reverse-proxy `basePath`, browser-language auto-detect, per-key device tracking, root-less MITM trust, zh-TW localization. → [Environment](docs/reference/ENVIRONMENT.md) -- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **348-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) +- **🤝 More providers & agents** — Cursor Cloud Agent, Grok Build (xAI) with browser + OAuth login, Ollama first-class card, Claude Opus 5 & Sonnet 5, Kimi official partnership (Code/Web/Moonshot), Zed, Requesty, SenseNova, Yuanbao, Agnes AI… and a refreshed **349-provider catalog**. → [Providers](docs/reference/PROVIDER_REFERENCE.md) - **📡 Routing transparency** — every response carries an `X-OmniRoute-Decision` header naming the strategy/provider/latency that served it, a new `cache-optimized` combo strategy + Auto-Combo `cacheAffinity` factor route repeat requests back to the connection holding the cached prefix, and a read-only `/v1/auto-combo/{channel}/candidates` endpoint exposes an `auto/*` channel's live candidate pool. → [Auto-Combo](docs/routing/AUTO-COMBO.md) - **⚡ Local performance & infra** — one-click local Redis, Cloudflare Workers / Deno Deploy relay deployers, Bifrost & Mux as supervised embedded services. → [Embedded Services](docs/frameworks/EMBEDDED-SERVICES.md) @@ -642,11 +642,11 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
-## 🌐 348 AI Providers — 90+ Free +## 🌐 349 AI Providers — 90+ Free
-> The most complete catalog of any open-source router: **348 providers**, **90+ with a free tier**, **56 free forever**. +> The most complete catalog of any open-source router: **349 providers**, **90+ with a free tier**, **56 free forever**.
@@ -990,11 +990,11 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ `:latest` follows the highest **published** stable SemVer. It does not track git `main`. Pin `:X.Y.Z` for GitOps. See [Docker Release Channels](docs/guides/DOCKER_GUIDE.md#release-channels).The image pins **`OMNIROUTE_MEMORY_MB=1024`**. That is enough for the dashboard and a light chat. **Coding agents** (`POST /v1/responses` from Claude Code, Codex, Grok, …) need a much larger V8 heap or the process `FATAL ERROR`s at ~12 GiB under two overlapping long contexts. Size the container above the heap (native buffers sit outside V8): -| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | -| --- | --- | --- | -| Dashboard / light chat | `1024` (image default) | ≥2 g | -| One coding agent | `8192` | ≥10 g | -| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | +| Workload | Heap (`-e OMNIROUTE_MEMORY_MB`) | Container (`--memory`) | +| ----------------------------------- | ------------------------------- | ---------------------- | +| Dashboard / light chat | `1024` (image default) | ≥2 g | +| One coding agent | `8192` | ≥10 g | +| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 g | ```bash docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ @@ -1003,6 +1003,7 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \ ``` Full table: [Docker Guide — runtime RAM](docs/guides/DOCKER_GUIDE.md#runtime-ram-for-coding-agents). + > **Pre-release Docker channel:** `diegosouzapw/omniroute:next` and > `diegosouzapw/omniroute:next-web` follow the current default `release/v*` > branch. These mutable tags are intended only for testing unreleased fixes and diff --git a/changelog.d/features/10987-logfare-free-provider.md b/changelog.d/features/10987-logfare-free-provider.md new file mode 100644 index 00000000000..507a528411f --- /dev/null +++ b/changelog.d/features/10987-logfare-free-provider.md @@ -0,0 +1 @@ +- **feat(providers):** add Logfare as a free OpenAI-compatible provider — dashboard card with a Free badge and request-logging disclosure (every prompt/completion is logged for research; opt out at logfare.ai/consent), live model discovery from `https://logfare.ai/v1/models` (20 models, 11 chat-capable: kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3…), full chat/streaming through the existing OpenAI-compatible path, the real Logfare logo on the card, and a listing in the free-tiers guide. ([#10987](https://github.com/diegosouzapw/OmniRoute/pull/10987)) diff --git a/docs/diagrams/cli-terminal.svg b/docs/diagrams/cli-terminal.svg index e2ad57c8b13..6ec68793d0c 100644 --- a/docs/diagrams/cli-terminal.svg +++ b/docs/diagrams/cli-terminal.svg @@ -1,4 +1,4 @@ - + Compact animated terminal cycling three real OmniRoute CLI commands with a typewriter effect and a scrolling subcommand ticker; the first frame shows the completed providers-list screen. diff --git a/docs/diagrams/comparison-table.svg b/docs/diagrams/comparison-table.svg index 053194678cc..f61e7ca0add 100644 --- a/docs/diagrams/comparison-table.svg +++ b/docs/diagrams/comparison-table.svg @@ -1,4 +1,4 @@ - + Static-header comparison table where each capability row fades in top to bottom; the OmniRoute column is highlighted and shows a check or a leading value in every row, while competitors show a mix of checks, partials and crosses. diff --git a/docs/diagrams/promise-pillars.svg b/docs/diagrams/promise-pillars.svg index 99b7f36b15a..aebefefabbf 100644 --- a/docs/diagrams/promise-pillars.svg +++ b/docs/diagrams/promise-pillars.svg @@ -1,4 +1,4 @@ - + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. @@ -21,7 +21,7 @@ - One endpoint. 348 providers. Never stop building — OmniRoute picks the cheapest one that works. + One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. @@ -38,7 +38,7 @@ Never hit limits - Auto-fallback across 348 providers in + Auto-fallback across 349 providers in milliseconds. Quota out? The next provider takes over — zero downtime. diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index 99543d2471e..3448cddc7ff 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -28,7 +28,7 @@ Never stop coding. - Every AI tool → 348 providers — 90+ free — through one endpoint. + Every AI tool → 349 providers — 90+ free — through one endpoint. Claude Code · Codex · Cursor · Cline · Copilot · Antigravity  →  FREE Claude / GPT / Gemini · auto-fallback diff --git a/docs/getting-started/FREE-TIERS-GUIDE.md b/docs/getting-started/FREE-TIERS-GUIDE.md index 008fe962577..6fd9dcc35bd 100644 --- a/docs/getting-started/FREE-TIERS-GUIDE.md +++ b/docs/getting-started/FREE-TIERS-GUIDE.md @@ -26,6 +26,7 @@ These providers have a recurring, keyless, or uncapped free-access path in the a | **Kiro AI** | Claude Sonnet 4.5, Haiku 4.5, DeepSeek V3.2, and others | Audited catalog estimates a 25K-token shared monthly pool | OAuth/account flow; ToS flagged `avoid` in the catalog | | **OpenCode Free** | Current `*-free` model set in the provider registry | Keyless; no published token cap | No provider credential; ToS flagged `avoid` | | **Pollinations** | Current keyless model set; some former models are discontinued or key-required | Keyless; no published token cap | No provider credential for the keyless models | +| **Logfare** | kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3, and more | Free API key (no rate limits, no card); **every request is logged** for research (opt out at logfare.ai/consent) | Instant key at logfare.ai/register; ToS/privacy at logfare.ai/tos and logfare.ai/privacy | | **Cloudflare AI** | Workers AI catalog | Audited pool estimates ~30M tokens/month from published usage units | Cloudflare account and API credentials | | **Gemini** | Gemini Flash family | Audited pool estimates ~60M tokens/month | Google AI Studio API key; rate limits apply | | **Groq** | Llama, GPT-OSS, and Qwen models | Audited pool estimates ~15M tokens/month | Groq API key; rate limits apply | diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt index 188e4c546fd..23e919c6280 100644 --- a/docs/i18n/ar/llm.txt +++ b/docs/i18n/ar/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/az/llm.txt b/docs/i18n/az/llm.txt index 1778044af16..fcb2eb38907 100644 --- a/docs/i18n/az/llm.txt +++ b/docs/i18n/az/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt index 1778044af16..fcb2eb38907 100644 --- a/docs/i18n/bg/llm.txt +++ b/docs/i18n/bg/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/bn/llm.txt b/docs/i18n/bn/llm.txt index e1f037d6698..f15efad72be 100644 --- a/docs/i18n/bn/llm.txt +++ b/docs/i18n/bn/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt index c181079187f..5c4acb63f9a 100644 --- a/docs/i18n/cs/llm.txt +++ b/docs/i18n/cs/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt index fea53e17b2a..e4ca44f80ca 100644 --- a/docs/i18n/da/llm.txt +++ b/docs/i18n/da/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt index 494914668a9..833ca47bf6b 100644 --- a/docs/i18n/de/llm.txt +++ b/docs/i18n/de/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt index 6dfe5a95b36..332a3a47890 100644 --- a/docs/i18n/es/llm.txt +++ b/docs/i18n/es/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fa/llm.txt b/docs/i18n/fa/llm.txt index 55c6e845de3..c6de54f2711 100644 --- a/docs/i18n/fa/llm.txt +++ b/docs/i18n/fa/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt index ca8eb2a75b3..dda61789985 100644 --- a/docs/i18n/fi/llm.txt +++ b/docs/i18n/fi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt index 6fc229f45d5..4dd7139b7e0 100644 --- a/docs/i18n/fr/llm.txt +++ b/docs/i18n/fr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/gu/llm.txt b/docs/i18n/gu/llm.txt index 0ef4d516e71..ecb0741ac4f 100644 --- a/docs/i18n/gu/llm.txt +++ b/docs/i18n/gu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt index e3d3b77ab8e..de5e856bac4 100644 --- a/docs/i18n/he/llm.txt +++ b/docs/i18n/he/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt index f92757a9f3f..2a6d15b8697 100644 --- a/docs/i18n/hi/llm.txt +++ b/docs/i18n/hi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt index e3ba1fa0a65..078d59d88ee 100644 --- a/docs/i18n/hu/llm.txt +++ b/docs/i18n/hu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt index 5a169f9e862..ba42452fc49 100644 --- a/docs/i18n/id/llm.txt +++ b/docs/i18n/id/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/in/llm.txt b/docs/i18n/in/llm.txt index e0b18cb0baa..79c9a604265 100644 --- a/docs/i18n/in/llm.txt +++ b/docs/i18n/in/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt index dbde86f284b..12f0f6555a4 100644 --- a/docs/i18n/it/llm.txt +++ b/docs/i18n/it/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt index 4bae4264258..87ca4d8a177 100644 --- a/docs/i18n/ja/llm.txt +++ b/docs/i18n/ja/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt index a2310374798..a6203810bed 100644 --- a/docs/i18n/ko/llm.txt +++ b/docs/i18n/ko/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/mr/llm.txt b/docs/i18n/mr/llm.txt index d40eb377976..dc8c256fa97 100644 --- a/docs/i18n/mr/llm.txt +++ b/docs/i18n/mr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt index cdf457c441e..9efa530fb7d 100644 --- a/docs/i18n/ms/llm.txt +++ b/docs/i18n/ms/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt index b2539aceb9e..838311afb37 100644 --- a/docs/i18n/nl/llm.txt +++ b/docs/i18n/nl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt index 9b4dedc1274..2e7e4f30e4b 100644 --- a/docs/i18n/no/llm.txt +++ b/docs/i18n/no/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt index d9b674c08eb..f40f7ff96c7 100644 --- a/docs/i18n/phi/llm.txt +++ b/docs/i18n/phi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt index 4c43c03d360..b8e216bf410 100644 --- a/docs/i18n/pl/llm.txt +++ b/docs/i18n/pl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt index ecbeb9ad4c8..92402048cc2 100644 --- a/docs/i18n/pt-BR/llm.txt +++ b/docs/i18n/pt-BR/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt index 7cd77fdd4a6..1d8ea55c53b 100644 --- a/docs/i18n/pt/llm.txt +++ b/docs/i18n/pt/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt index a4f03e94e7a..cb8a15c28f7 100644 --- a/docs/i18n/ro/llm.txt +++ b/docs/i18n/ro/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt index 8fe9b5bdb68..1a1c527403d 100644 --- a/docs/i18n/ru/llm.txt +++ b/docs/i18n/ru/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt index c5a67d9242e..9ecbf58571b 100644 --- a/docs/i18n/sk/llm.txt +++ b/docs/i18n/sk/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt index 3b7b7b674ea..e3e56731c6b 100644 --- a/docs/i18n/sv/llm.txt +++ b/docs/i18n/sv/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/sw/llm.txt b/docs/i18n/sw/llm.txt index 4c7bec4311b..9f19d4a92ef 100644 --- a/docs/i18n/sw/llm.txt +++ b/docs/i18n/sw/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ta/llm.txt b/docs/i18n/ta/llm.txt index 5323983d315..952ef1584c2 100644 --- a/docs/i18n/ta/llm.txt +++ b/docs/i18n/ta/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/te/llm.txt b/docs/i18n/te/llm.txt index ce196f99275..7ca0ad8bc91 100644 --- a/docs/i18n/te/llm.txt +++ b/docs/i18n/te/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt index 39eb9c04ff8..fa1c6d5ba90 100644 --- a/docs/i18n/th/llm.txt +++ b/docs/i18n/th/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt index 7f848b6cdd6..72ffb228ff9 100644 --- a/docs/i18n/tr/llm.txt +++ b/docs/i18n/tr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt index bca18382ad2..f2fe37d1f85 100644 --- a/docs/i18n/uk-UA/llm.txt +++ b/docs/i18n/uk-UA/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/ur/llm.txt b/docs/i18n/ur/llm.txt index 85369bee2e9..e731f3ccf57 100644 --- a/docs/i18n/ur/llm.txt +++ b/docs/i18n/ur/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt index da6b9b7179a..38e56d30c5e 100644 --- a/docs/i18n/vi/llm.txt +++ b/docs/i18n/vi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt index 0c737e5dc43..b3ca89b4d17 100644 --- a/docs/i18n/zh-CN/llm.txt +++ b/docs/i18n/zh-CN/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/i18n/zh-TW/llm.txt b/docs/i18n/zh-TW/llm.txt index 643943547df..a9c38ef389e 100644 --- a/docs/i18n/zh-TW/llm.txt +++ b/docs/i18n/zh-TW/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -169,7 +169,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -479,7 +479,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index 8a0a3b5d99a..571fe0e904c 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -1,16 +1,16 @@ --- title: "Provider Reference" version: 3.8.50 -lastUpdated: 2026-08-22 +lastUpdated: 2026-08-21 --- # Provider Reference > **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. > Regenerate with: `npm run gen:provider-reference` -> **Last generated:** 2026-08-22 +> **Last generated:** 2026-08-21 -Total providers: **348**. See category breakdown below. +Total providers: **349**. See category breakdown below. ## Categories @@ -34,7 +34,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each --- -## No-auth Providers (no key required) (12) +## No-auth Providers (no key required) (11) | ID | Alias | Name | Tags | Website | Notes | Tool calling | |----|-------|------|------|---------|-------|--------------| @@ -47,7 +47,6 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `felo-web` | `felo` | Felo | No-auth | [link](https://felo.ai) | No credentials required — Felo is a free, no-signup chat/search aggregator. | — | | `opencode` | `oc` | OpenCode Free | No-auth | [link](https://opencode.ai) | No API key required — uses OpenCode's public free endpoint. | — | | `theoldllm` | `tllm` | The Old LLM (Free) | No-auth | [link](https://theoldllm.vercel.app) | No credentials required. The executor auto-generates access tokens via an embedded Playwright browser instance. | — | -| `uncloseai` | `unc` | UncloseAI | No-auth | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | — | | `veoaifree-web` | `veo-free` | Veo AI Free | No-auth, video | [link](https://veoaifree.com) | No auth required. Rate limited to 6 requests/hour per IP. | — | | `zcode` | `zc` | ZCode (GLM Coding Plan) | No-auth | [link](https://zcode.z.ai) | No API key stored by OmniRoute. The local ZCode app-server uses the existing builtin:zai-coding-plan login. | — | @@ -99,7 +98,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | — | | `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | emulated | | `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | — | -| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://chat.minimax.io) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | +| `hailuo-web` | `hailuo-web` | Hailuo Web (MiniMax) | Web cookie | [link](https://hailuo.ai) | Open hailuo.ai, log in, then open DevTools → Application → Local Storage → copy the "_token" value. device_id/uuid fingerprint fields are derived automatically; if requests fail, re-capture _token (sessions can expire). | — | | `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | — | | `hyperagent` | `ha` | HyperAgent (Unofficial/Experimental) | Web cookie | [link](https://hyperagent.com) | Paste the full Cookie header from hyperagent.com (DevTools → Network → any request → Request Headers → Cookie). Session cookies power chat + billing usage. | — | | `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | emulated | @@ -121,7 +120,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | | `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | -## API Key Providers (paid / paid-with-free-credits) (231) +## API Key Providers (paid / paid-with-free-credits) (233) | ID | Alias | Name | Tags | Website | Notes | |----|-------|------|------|---------|-------| @@ -150,7 +149,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | | `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | | `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | -| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | ⚠️ **DEPRECATED.** api.blackbox.ai returns HTTP 404 on every path variant (sweep 2026-08-21); the public inference surface has moved to the gated enterprise.blackbox.ai/v1 endpoint. | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Limited free access is available through Blackbox; model availability and account limits apply | | `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | | `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | | `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | @@ -167,7 +166,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `clova-studio` | `clova` | Naver CLOVA Studio | API key | [link](https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary) | — | | `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | | `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | -| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /provider/v1/chat/completions endpoint. | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | | `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | | `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | | `cursor-api` | `cua` | Cursor API | API key | [link](https://cursor.com/dashboard/api) | Paste a Cursor user API key (crsr_...) from cursor.com/dashboard/api. OmniRoute exchanges it for a session token on demand; no IDE or cursor-agent install is needed. Usage bills to the Cursor plan that owns the key. | @@ -214,7 +213,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | | `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | | `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | -| `hackclub` | `hc` | Hackclub AI | API key | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | | `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | | `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | | `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | @@ -243,6 +242,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `llm-kiwi` | `llmkiwi` | LLM.Kiwi | API key, aggregator | [link](https://llm.kiwi) | Free plan exposes auto and hrLLM; the published 40 requests/hour limit applies to hrLLM. | | `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | Use any non-empty key (for example 'unused'). If older built-in models return model_unavailable, use Available Models → Import from /models or Auto-Sync; verified live model: gemini-3.1-flash-lite. | | `llmgateway` | `llmgateway` | LLM Gateway | API key, aggregator | [link](https://llmgateway.io) | Hosted Free plan: free-priced models are limited to 5 requests per 10 minutes when the account has no credits. | +| `logfare` | `logfare` | Logfare | API key, aggregator | [link](https://logfare.ai) | Create a free account at https://logfare.ai/register (username/password, no email verification) to get an instant API key, then paste it here as a Bearer token. | | `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | | `magnific` | `freepik` | Magnific | API key, image | [link](https://www.magnific.com) | Get an API key at magnific.com/user/api-keys (header x-magnific-api-key). Legacy Freepik developer keys still work. | | `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | @@ -332,6 +332,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | | `typhoon` | `typhoon` | Typhoon | API key | [link](https://docs.opentyphoon.ai) | Free API key with a 5 req/s and 200 req/m rate limit. | | `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. If older built-in models return 404, use Available Models → Import from /models or Auto-Sync; verified live model: solidrust/Hermes-3-Llama-3.1-8B-AWQ. | | `unorouter` | `unorouter` | UnoRouter | API key, aggregator | [link](https://unorouter.ai) | Models with the :free suffix do not debit balance; limit is 1 request/minute per free model per user. | | `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | | `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | diff --git a/llm.txt b/llm.txt index 51feb87bc32..ac68d5cb4e4 100644 --- a/llm.txt +++ b/llm.txt @@ -1,6 +1,6 @@ # OmniRoute -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 348 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 349 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -165,7 +165,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ └── manager.ts # MITM proxy manager │ ├── shared/ # Shared utilities, components, and constants │ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) -│ │ ├── constants/ # Provider definitions (348), model lists, pricing, routing strategies, MCP scopes +│ │ ├── constants/ # Provider definitions (349), model lists, pricing, routing strategies, MCP scopes │ │ ├── contracts/ # Shared API contracts │ │ ├── hooks/ # React hooks │ │ ├── middleware/ # Shared middleware utilities @@ -277,7 +277,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **348 AI providers** with automatic format translation +- **349 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **18 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free @@ -475,7 +475,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool ## v3.8.x Highlights -- **348-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add +- **349-provider catalog** with 90+ free tiers, one-click account imports, and bulk key add - **19 routing strategies** — including `fusion` (parallel panel + judge synthesis), `pipeline`, `reset-aware`, `reset-window`, `headroom`, and `context-relay` - **14-factor Auto-Combo scoring** with bandit exploration and progressive cooldown - **MCP server expanded to 110 tools / 33 scopes** (canonical + memory/skill/agentSkill/githubSkill/pool/notion/obsidian/localCorpus/gamification/plugin modules) diff --git a/open-sse/config/providers/index.ts b/open-sse/config/providers/index.ts index b0135fa5242..3e2c812e851 100644 --- a/open-sse/config/providers/index.ts +++ b/open-sse/config/providers/index.ts @@ -264,6 +264,7 @@ import { freeAiProvider } from "./registry/free-ai/index.ts"; import { voidAiProvider } from "./registry/void-ai/index.ts"; import { helixmindProvider } from "./registry/helixmind/index.ts"; import { tabitokenProvider } from "./registry/tabitoken/index.ts"; +import { logfareProvider } from "./registry/logfare/index.ts"; export const REGISTRY: Record = { aimlapi: aimlapiProvider, @@ -532,4 +533,5 @@ export const REGISTRY: Record = { "void-ai": voidAiProvider, helixmind: helixmindProvider, tabitoken: tabitokenProvider, + logfare: logfareProvider, }; diff --git a/open-sse/config/providers/registry/logfare/index.ts b/open-sse/config/providers/registry/logfare/index.ts new file mode 100644 index 00000000000..9b16b5a2f4e --- /dev/null +++ b/open-sse/config/providers/registry/logfare/index.ts @@ -0,0 +1,25 @@ +import type { RegistryEntry } from "../../shared.ts"; +import { buildOpenAiCompatibleRegistryEntry } from "../../shared.ts"; + +/** + * Logfare — free OpenAI-compatible LLM inference provider. + * + * Live-verified 2026-08-21: GET https://logfare.ai/v1/models returns a real + * catalog (20 models; 11 chat-capable incl. kimi-k3, deepseek-v4-pro, + * glm-5.2, gpt-5.6-luna, minimax-m3). Auth is a Bearer API key issued + * instantly at https://logfare.ai/register (username/password, no email). + * + * ⚠️ Privacy: in exchange for free inference, Logfare logs every request + * (prompts, completions, metadata). After PII scrubbing this may feed their + * private internal evaluation datasets. Users can opt out at /consent; see + * https://logfare.ai/tos and https://logfare.ai/privacy. The dashboard card + * surfaces this via freeNote. + */ +export const logfareProvider: RegistryEntry = buildOpenAiCompatibleRegistryEntry({ + id: "logfare", + alias: "logfare", + baseUrl: "https://logfare.ai/v1/chat/completions", + modelsUrl: "https://logfare.ai/v1/models", + models: [], + passthroughModels: true, +}); diff --git a/package.json b/package.json index 55cc574f9ec..526c60cbb78 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "omniroute", "version": "3.8.50", - "description": "Unified AI router with 348 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", + "description": "Unified AI router with 349 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { "omniroute": "bin/omniroute.mjs", diff --git a/promise-pillars.svg b/promise-pillars.svg new file mode 100644 index 00000000000..aebefefabbf --- /dev/null +++ b/promise-pillars.svg @@ -0,0 +1,139 @@ + + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. + + + + + + + + + + + + + + + + + + THE PROMISE + + + + One endpoint. 349 providers. Never stop building — OmniRoute picks the cheapest one that works. + + + + + + + + + + + + + + + + Never hit limits + Auto-fallback across 349 providers in + milliseconds. Quota out? The next provider + takes over — zero downtime. + + + + + + + + + + + + + + + Save up to 95% tokens + RTK + Caveman stacked compression cuts + 15–95% of eligible tokens — ~89% average + on tool-heavy sessions. + + + + + + + + + + + + + + $0 to start + 90+ providers with a free tier, 56 free + forever — Qoder, Pollinations, Cloudflare, + SiliconFlow… No card needed. + + + + + + + + + + + + + + + Every tool works + 33 coding agents — Claude Code, Codex, + Cursor, Cline, Copilot, Antigravity — + through one config. + + + + + + + + + + + + + + One endpoint + OpenAI ↔ Claude ↔ Gemini ↔ Responses API + translation. Point any tool at /v1 — + it just works. + + + + + + + + + + + + + + Production-grade + Circuit breakers, TLS stealth, MCP (110 + tools), A2A, memory, guardrails, evals — + 25,000+ tests. + + + + + + $ npm i -g omniroute  ·  point your tool at http://localhost:20128/v1  ·  $0 + MIT · OPEN SOURCE + + diff --git a/public/providers/logfare.png b/public/providers/logfare.png new file mode 100644 index 0000000000000000000000000000000000000000..223f6e39cdc5648fbfb2f86923614fac3c799b9b GIT binary patch literal 17858 zcmXtAbyOAI+dcQv-AK1INOvhA9n#XJbcggMq$NeVQ@T4YCEY6BAT1yvNd0)%x4t!N z=KM7?XXZTf#NPXvNOjfMSm$?Gff&bqQ znOOiS)^qQOm20rdreQZ6>2Yi-Y+E z7j5F$GtUHEw7yVlXMD`Ty&&?K>VdE3rE|1h9E#sU8N!IK>QQuKttmSyave@b@~?tC zepkTxA2Q*WR+I>VRCB^Lt@%o(b}Do5CWRFQ(TM}1dGVdk@UDG-=-152mMS&qG2cIVkacTo!7k*ha&{jZL`eE`FA2(Hs_=eQch-M_Weceg)e zS&u8|Jk+qJSNljC7w6xT-PM!ji~pN98kZiJyC>LhA1~r-ZWjNX!vHld!Dm$^(E9$W z^fzACjzD(=R%x&Wb{CvURkaIO%W#QyMR%>eDNQIZEj+5 z>lze!r7oveE|^g`b3{pVX)rY17cZPcIVnDqmDn|}r&AxIl@5n>6e_`Nj zzCRzyg+s99lUD)!aJUz~7MhwY%KKABPy0m0j4Vc|#kdgK% zxu-7xk$Amx6yufNr!xRgYU~WQey#nzl%b3_4*mFq3lrXt07uNvN_L-~`odS5muuIN zc5#u<`K{kVEC#|gwm-Bc$=%H>-dKtuE>O*TEL0!jW0l7%)M-!;lcdK))=gIz$3Bal ztTNnbjKS|G`;y}r9?{E;a`DVOsO5VLxT|w1zD0@Sz&zm?pfW> ztKF5G5MkaVrl~6l-4YTwrb7jTyI#dc@85qtjN%P;^Wm4&`;vwWC5bYqu~J|0JWH6P z6^N;go+w6ge&vL5w^1v(`Kx6Hkyhe$1Thv6zIj|03Zu$NdHq*yGT*nvBlouW{q#L= z-F{Ho!o~U+g^#rkr`mFXo9Ou(rQZCfY@?Z^RTJJ+s8quXbb zFxsbskhsRZFHoyz4BG<3kiY1@_)Wg`thD%K@;FP5J{_&?Ye(Y-A{9C?hvyoE4gT_c zhw^6_DQkaqU-+Q7PKVZ>;3Es&whF;^ZR8}nri@c7YJcRN^HRTw(br&O1MXfZ*g{jt zi>oG^Ah?w7yU%!%?XAI&9?`+t;qd*({h#>%9*C*$oulI<406=sCQo|5Y?{mw z*)9cj5pjVgy99U)6$@WkYwq~g(jE}%7>;{SS}5CVXWri8ZD`k6EL~tCZWHW>eOAHZ z<-F;LLY46WMd4YYEeE8gAZk_LLj++xge@HJxZ#Qba}cMXDpAXCMO)g*NJPr_H<2`{ zGBPkVYSL9djdkHGQo`S1!dQH>@J0L~Jdg=gup_`v$2K_|-YW_uTQVZz-CR4EHwd$M zzLI$IDeqjOa#J{|h!geUdmQV@(f>}fA8*2-7DTUc=#{_cFeQndR8b9q>!lrOIt~~% zuta3?u$L<^KyB-6ThTy)dDC?GxthIC_p)^bISVJaN-j7T%A1il&P&O2>kOUF8&kM% z>Zi-e8Od(%5T5L4O|Y#u$r>B&kk!27j^yyHB(D~fmgkx@eb)2{XVgk%wKj0@^)Y|B zYk>?pIGYd(Gw^7iA{kK&3X=#hMh5DQya-ZD1#EkB999?o!JK{wc)y14seSJyt0`3; z+o~c0I%IEK-{Js?4)fIKAfT-==3BFa3^U@7tm&YwI}0q>bA&hZ(Cf?*#pMtJa)7;q z)IX;7f|LRctlahGg}laj@woo8b`kNUG{3u=^5t!D!xk5=;NNOHp{#hHqCRvDchJfkh+?IMa637;DQpYV?;sOn5 zcmKl7uQqc;yLDxZ*UJyl_XjY1XEB~#6o>D+TT*YF*u9p!M#Y3FZ9z=M5Y{uu{k4A9 zLA)t~uaqa=Io<0an5tI)dI{iLguE^|s_sBcJC%J8o*z^T=JjQBs;CE&ZWsRqq^AiM zc`VFUmlt-M@CS4B)bIX{sRd4l0KTTq2R!L|wu=*E5dRdhH3ZsXGr7k;2Q>rAhd-5} zrZbTgNY#V!=o z&HM(|(f9_=eqB@F1Tq-LBZazI%OGLGAeYeff^u~4N9rZk1aC03Ny!FG#YCD3RK{#5 zUj6RdBK-Jd2m*j*q>&F8~!o0sY>3}E}ie(&lLf}i(sDhUP+1jfD|gk*Gs+-HO6 zY)LE67%b5$azxDqP9@sQ_U|P>73cH4Bv|CjMR`Z(Rzl<-JG`A!{`_NEJ=w?jdw0tL^u)8UVlDg{LR+_1XaS%^%eZ!B`fg~oxmXd9}oRw`blf)b1 z3V*f!%RF)8bXcAi6E@2SD#)!MFfyN;{0+Xkv|L_M0pWM6-H71Tn0QkCk6I1_0Lu|4 zr5p8op{fOr%{0TQ-y;BzY7jXR@9o*XP51%7*lUu)Sem6LFmmv*5 zP7*0KM&bR~b;^Wsi)Mf=z%JERM+xTrRh^UoNm3-=iv#eEGnc|=nHmG8=(Cmh@MK)I zNF}pyOop=#W7~G`Qv_k<~qFp_vm+2!Q20T}n!UWAw@RYfd zq5#JH*NFSl*jZTiO{g`vqUC!j97RV$bWzu~77XX>Gl$}w?esP)3(`^B4r8bwZA_S;HT=vlouU7pKd4Fq$bDGspGi zFE6s%!v(vKt%eu+a$ZS>w!v}g7k84U#8|AbzIrJq3NLC89#$mI)1?2K(JGS%;u?chnao@bMBXe zi30@$e8eoTwWXWA>E1xS~`mjXuhE>8^2szIVzYG zj42K@7L_BPi-{#M>5qX7k}GBzGP>Q7`^;e9AAJ9ll8Wi zOd>Xo^QO&o#<}TH;{9xQo`Jq-_=De+6FKc1)UHT{i?D79d1DBk=%U} zx*CjeNT7Gz+7obC=ZD0YG=D0wS$R8k${1XAIowoZF@*YbEsTFnrmvV`PF*;UYZB}I z6#4Xhu=qeH+bmUP7uGm4lgytS{MQ4Gj`{us3ZHbq0zvtOS&R{hw&9xKw+TfDlDs@B z0Xls6vrR zaiHVHlo-TfIuad`cMrEf!&7eH?po zJd0YJJ;}bqqqI!fU3Z(25qhfj59|jLZ&yuVTY&W7cF$LKiF?-G#iO+*F~ckdAg)5F&bYA zBMyOA+kloNQiT*uFJh-b}B;j>u++Sa-mBl`An$i zr_wTii?U(cR|Ib9Gij6=v-(--nhJ7s-HfSHRvZM@8jr-8D0<(kk-|fEapd!gN=l}X zLf+UctZTJ>kSF7ywk;s)b7*i4{lrOWQ}UyLqM;c6GLY=JknNk$ep$YITboeG&vxED zGBw7KgSSqMm47X#5@+8`VHL(I;FV1dd%#+$N`^d^WR3Gw8d#j@?U6|HZz>F@=WUg) zO$gK0qg{$3cT)MjpARzFQDdVHDk2ZlQzwN-+L!qcOtv_XeVyVH2-&X3V0A(LBTAGl zWg>FzjkY}~kD|{T6!_bNdApWPE1vxWz#I^`D!%y(LJ%ivii5)hhz++lruA+grx+L4 z)?2L+ArV}F4~?rs06__5k;E_i^L)_S)(JN#-cad7IEnk)1Haw03wp@5cq3EFrG5AI zGb4@SrIJhfb&nOnTATeQ8#`6AQOAx8D*G1{aZi^bQ56fU`$>B&*PV6;a~l>$Nqxkh zF^7hc*K1 zxUx(EayL#Bk{)pfC-``17HwIln5wQmT=e7XxUI|&O0t>$OgZ;=gWvdstI<@B7eBvq zclw5~wzW%>t<}`$%lJmWSWfP;1Yv!de+)yKGk=pX?^WVAD}r|Pq)^Qq=+3oTm(ljk z(MVisrl?(ecs$%}pI^wtdC2o{BK@<&l1aI+orw0~T;)y&b>9X8(@FA9-#&aZC^Kbn zdxfisz(`|0o0B3_pv1F|qB2brL?{+#VunIm0V4eoPx{UHI<5p)eSyzojNxUrjqt(mVf}+x zUXfC-pADMLI!#zBJXIdnn)>DIKdLU)?=+L3RN?ri7QXe-gUWThBWIshDAAPgMmdjm zQLTGb7gdtADp4bwMY+E1Z=`=sdPXJO;iinqh?*^7ce`)geQ|oox)*J?>^>YJ5cajx zGW8QD>APLT38WAnA*fUwSY>)erTW;7Nu*t`71LT8dMIUS3R*<);Tq{7tRTRP=>GB# zRHWUFCj>0@RuU}seZXQ$$ z^9dI$$|59Y1>GiEe*?kRXZ0WQlLRlZ7CPB%40O_A(`Vw2O+H>kMtp_OGfdt8Xf-hSfue-xYAv)gQgNu!Tk6m75JJI(nn_wR^LxITCMPKQU*R*ZR7lQ>)>&}f@J z`aec-W5!=^zS*EcNMzYWg?t?%b`x1XTogd~f@pFDbE!NN&F#kTI8fw=Y_c{lPsvz;i6Muv z@+)hNQDrUH#xiMXq6r&?;ytg9CWy3ZmX*;AVrLhR&fFk#(=vZhxyMN*mCQ*sX>X@D z?gy#*T`xi3{apj4mN(C)4{=~aO;x4ZF*%QQjhR~?01!l|70(nt7XFzOTDdV!gs_*l z&}#T9!1h%UnAV@ON`eG>w9l3izXwO5B%86E8Q0K*#lUF#-w}^nzw)W#nmC2k+fmAOb(XUY zzkia=$pz4dhu6B87zAP#K42Ou@Wr-1!<7kV^R%_oAwj1%nBaQVDr9ff=6GSl=cxu6 z!;zxSyIBO74Y>vFsFZ^(UtS0hI`^TuhguulgO;Lv3%_mc3 zWeiHiD;lR?7@&OlaQl~07ZG*L9GmEvtoG)$UW?xcw(*w@JdevfIulq0dn8abn5UKt zIjPju<)!R05B5Fm*^50DwXr$H9uOkOIGmsI;-Em-cj*uXo;;w>&L>>p>sHx$$IKgf zB2h0bKZ)X}pX#0Y2ZH=*iN!Taa{$M}fN>^J;O0l_y_YYN8lNM>-|dsK=RZ0I9YfBw zCTyu6JBLRJAZM@b@#-*8$En_x(+yB`b+>-Ub(2D@Sv1_@!Jh zT|g6lCplX=+mS}A^s2$;5i5d<+3{I82+Z~`9$sy;0T`0M0fn|sLYe|5w+itOdzN%c zUGP~*wwn7Z9jZ|lwLr?(x8yn>{?t}Wk0US=5yY&8CR5JQ7B)x zlC|^_9F){@kaGb!KB9)kVp07o{Jk%G^hUnI)rsvlJw3euOrHrwn;gbudTz__x&J({sVHoAE+Dg^`ZZrJf zRzKWvE{~Z}x(=uY$Mjk8GH;vEgPQupu8xU576SkbK-tGH>~t}0ii}}ol^Z^Gtz=A# zsJqXlmiD5}{Kg;9oF7|SV<kadeBTo_yM#nyxKWAIUd}WKCFxE)K)^)^VL)QFzZa2zTh9Oz9RLf> zyDuodndvDL=AS&)$SiZ;y|1*;8|WX6=mRMq&s#KSX@Wyg!>3rsKIh)eVM(w$t-$I( zlViOZo<;!1PMxCCe^5{6O6WIQih`7Q2u5l5>#q6UOa1!!32jAVz>LU|LeMtdk)q9Q zgFveGoQ8nEpjTXZ@n@~Q@aYC#`RJTrKgCaC*FPvgiW!wks-d^V`*Coae#Z0%uK&`_ z19462#@G{zD;@>O!wxqE`NpM+v#Wt0Xfw?rFjP%F$n4TOVaGaFsI~yC8`w83>&^l~|&) zdVeZ*4Nj3NJ}VS<0rc1Mv-yVpbNvP7yr@9R$bARZ07Oq5YB>ryHUbNbA$h$B$JNon z28@1Gk>#(^A`@f9PLAw9#T@Q;Pb*AJR&%nAzi0Jz?oOPLB2ZPRR)!RpPY>a7{DS>J zrnWS9VZycy-|N34nmd-Wg5yrUJCg9GKAae8O^OsW4dF z7&+X^{IeG=f-j%;w{2qQ9{6{_4klFiUVPhvf+vm$9kl$+u>ex5?y1_#dNDQXArjma zp4x8L;kX#bE06EuLzz$(gk%5(a{8td8f;D{X8lLpeW2W=)*)ur^5ZkJ$4PEN>6pyk zCrsn4aGQ6()i>I>0M`Xd^UU_;>BXDpj+H)X1hIL-H|4+vyo9gG* z&AG2^h0yRu;C{YYCKMS5vW=}O{W{~ZO)-|H=FuWJ;hH;#C|7aN>Gg;9f9^(IjISys zsigueITag>^n%3xRY3ZnN99E-TYp+`-92|1GabaP3zd*S05!nt0I&ZWcgIOr!_oJMBl5=pC#)pCb$eRG zjxr%OMkwmZBmE~gsL;O%3l`(q>JZsdEE=ekql%Yvq+=v|_KV7y9P}kOd~4O}LW_JJ z^c-v3x#G8?X+~OmJu46@@#$XOI)ce$^uuK8JE)(a52x$#l%c zDmKZ?L%(3k3%?sLdcSR3=Nb_1oUyx!<3BXjcc_jH!#EACIwC~|{Um%Uk~-UqMoML^ z{sx}ER*f7|jg?!xG2*jt$=UuHG`ULh1cHpfDfpGz34tIMCN-30J<#G&(27o5b2;v> z%-k6QuUpG28{~YkTqwSq4t6dn?reLBGqLjy>MgeoM_;W+@-cZSCTxqDjfqzcop}|& z*`DFxC=%Eg8dpxSZ`^9zu;G_uP4~ho**0jNb zTC?pB1xv^ramiq&kSAXY_&YwNJ!n&8Iy1>A)>UNc{6Kp$QR0`@3VbL*xnk410^}F< z=6>sJ}!kIk10<`f`dV=lT?=Aty;juYdXTWCD;B&BK`?qmJ zScEgqM0U2_Tsqg_K8i>(If}>zK;74qJw<)WBQ8vf4p<;H_hZ>d&UyHebjt4dLx9fw zMs&n`$AbE`HyPD^(8u&l_Av?gQ+kBJ(xQU?M#dGD3&L!#zXL3_f@=uz&^~|j%JO%g zB;!g6*4G~6e9*8a;D{A{mdmA$_;Td>*Ec;*Seib0)-F!Yudi%AX*pggH}*84RyR** zw}Eu?oaY>2r%$o1q%`fwjP%@6Z&7-x<7(i_n{ThDxSxNr!!pBJ^OMOc-k%$&c{6=N zTfO_@9GQ#A-Y)~$eX^l|jsFGzN9PFs@+jec10kEs7HBSerXj0NQ&?6 z!Y{e2bVjb)vj1NT5J`)Gxx|(li{wV>-=AUVvUeAhIr>zL9HLEDqzxhC9x7`@DoDaJ z$AS@TlE3|8v?mm;6T;{sWP>U+2h#}>R5Z2PArmyf(>5^V+TBtW=7T{Wnm>orJ8Kbf)s9og$;je#h zE1!`&M6cKLcS*8Qd~n^lW!-8Ep$8Jm8=?Ji6GK@%iht@XIrF8E12QE$>n-EQtd|f3 z;Ful%ei+gG;M(Cig-`Epd?ZsybdBa)rt^)N1ey4!$Xu#3A;(M>j?M_F5$Ue{i8IA? zQ{xRlv{;_lTPM(k_+3WeA5ZPy@&Hl6Y3#Y({Dkj%TsFKE21bxHA9fYBQkmJ!-iHa? zUYdHe0n?MmJL4nSkZ_du=*E)D^b#a6h?8NW6xO#>zE5e<)6+x3n~5(8=-!WE9r)tqQ-UO_DH0GuOj*vE<*E z5Mq^wFXIt%S7`uU)-Vw+U+||7>nk*cuW8%nex$}ZC(oi^t1JuEg(fV1>>gp4T zFWhQ|{qc`Yxjn#mq_kd17Ih4xvdUXk=<%uz<%htO&-{Rh0GmZb`Fg1w`Sc{dTgr;Z zQCwQm8UOVJ-HY1d+yGX+>-hcHZ)fM%kKZ#F2C7%GHvw`tsYGo-xLU z+SV*RE5u^|!8A(iAM&Ej3G@YMOfyq2ruZ@SP^IT+n5}8G8X`joB7#11vF6L#(3Kuy z_@ShCjWRx<{j3>fTs@l~fK+SR{}oAYq-sAM;Goh^LusICPm~?Z=hN6$B#3@mu<$3? z&%G$}aq|<#TV_^NT+R04sX4)asGG5@7v57wY)jSjWUK|E?uXRef`*S|Ef!)bCkcvu zWEeg-iFhs$+^u7B-@s(-co%z{nqE8yIC^K`jp?dLe6BfEK66;miQmNWq9y`~ai4fL zZ`QTZj?q|D_HNZixPpYI{x$D}xp`ubEpnu#%OmmfpI8UWeYn=cV~1vvwddq7zD@() zp^@D06e9H&TkHKtayy)dUL=ZL76hVODc*6S3W9HcBuLg1MU?e zfYy;CH92TS@SG~#wRMp5_tSe=d!lQT*78`qOr?+z#ZRA@rn0H;S)XSk*j#_;k6Z69 zlsK!2le9}Zau!V)r4KqbXj;dE8WhnffZ!LgSTjbyg^qqk3KhWH-KCt8x1Lm{sqpms$PYt$Dn_v|^l4J!WJ8F4 z*Rk+b8fd=VomjUxh|zI2iXN&^bC+r+E*jBt-h>l3Wo(F3nlIwE3gTfEI&&6pZDJqC! z*m%yOpH1~gWjt(Cidp&Gi?{i$7@jQ=PK4%ZqPRO)pN^f_l@QS{O+AT+_J0qTtV_uw zqxREGEe;4$`(FcOM2f3DkU%Vx;*vxUPVTxC+)s`k4ADaAP!Z`@^);p3PZ-d93g(y? zi<3oFnd)qjVx^DpztHaUDuvg44Wq&xatOBS6&gk7fiWA;gjf z1@fdsM^@M5LblGa+b5@FAH=mLeq|=grNr8*2*w`w18?(bsbBwkg74tOme6oNfT}cb zTxMQ`ZPMmBfvrqs>|}WOP3V2Z!})v3TNLEdF(^UZf>t#?vYCsr&fu7Ksnd+bE4V+u z`$>9UUvxt(M@(K&kGtBx zFcFK1)9I1P#WwMJr?&WJ<&z^-&g(<*g$7$K@-}O0CC1n665|TIKEda7F1eL*!ln=o<9)YXc42>!xCrovgj4gQ|w@3-?Y+Ltn~^;Vs53;rX!KGv3bE z=-q;a1~lB~^TCjdZzCGZuVIM}^9SE;ZK>@&=w|z;dvUp==-MQEO_xlR4m*U!rimPA zb*5p}NH8Ejxjz&7lenm!GuJSIr^k(b@_Ahqll5nwYozD+!y!(E>rrAJ7_}`DH$hqULUSyjV>7PvIT~eIpO<2OW!o0~ zxN#uSiVq)b3RR?PK>A=19ASB}J1wasRPh5AH2ipaV9q*f4?hXAlh#C)0>Z7Lc2dG8 z{pFO5Z_}?Vyh*J5No5TOqc+kdYA5rS(Y;f!!UUcOO><8;_G`}Pav}o_wtAo39%b5W z45_MP(ktZrG{WHd6WM8rbD>Pl%-VV9$U-(23Y;W^V~qyFWmp{Ki3%9EqSmS>d|y1o z>owh~kYnvuVC|uL64-fPRWjV|ag@pJiXF6P7B6%Wv%ci&)4=d)SjXiH>qVKS^Jg=2 zTkBE3-`D?hNj~W-N7pxeQT<3>!X!hIie6Z4hL zuN@&O0OxnOXv7Te+wh1$pz)AE)Hw+`cYgDpz2KYm0MFprv(ZcYwp{Qv&Ofe78IG2!uz&8Bwwu z_ixousPoj+B|$HdEPz`)b6Rists6h#?=!rvj`?*)D_!;sdH>DDVeN>nSaRj@h66p7 z+pk_)jvGe$iJhr;#_ty;YQ4C5zwIhhx0ubY9_`4h(ad2~D{RxVBzeDDMfd!m*SMR)eT~7#eM72G@TGjTcH8+F zHv2_{0;j4{ZoP=$E(c0hccZ?F^{b4qqf+xj_!``#=ZX%+j|kH*Zt{6+^nCg{2>Gc! zT?NUSuI0{}C7M>pH;}4{OuoO3ME8&rc>jIO!$*HGW%*`*rLpeo@6^+5 zzx#RBGo|*|-labH=&C0%RA0qjK20Gezy7^C{Yy91+UccoIh$1pjMp(mPELPjB?a5| zw;DfQ{I)`(>F1H9eCT}wBe>Nd(7BXPj-%7Ke#A12&Qi42kmz6sl#{_O1q&>89{gU| z$CvY{QZz<)^Brp^g7VbsBHb>k(-1vaGX%E#g2a~DPk{$9*0QJHd?2JV`ZJ5QhI~f8 zT-@@%xw&V&J@^jmL-o>1C`cq9S^+Ay`{xBd=~qRi2-PtQor##Tj%Z*T!JJ*=yMXH+ zeWvhTrOZSTl7lzT`Veh@e^)HgBV|5xn+CJIw7F^Y_*=*LkXeMnPwT)zeAk@|`n%)| z#l7-Fv5I#OKa1%1Z!p5A3NVnTo(gDZb>fsJye`#~)nIxO6%_7-E7jqDEB;|T=z0bf z)S6p_e|w=F(NU`y&*UJcPf+Ho0^6(KRw1c{c_TPZRHFIlLOvR=siv+IBBxj&&f2xe zPnhsq=O%Oj&mJbuttfa7=&X(E;w`mhc&O<}$FIKbmZp}kEZ7$p;qqVIF-h^yVHi%^ z(&hExCi1PSx#Eqn7UGM&`KE)DC%wcpmHoAIcuU?Cw}Q-jFF&ap3HHR}zNC1w;lXSW zI#_i_)KK*-Mb-c|GBiQeX>;DJlg03+Bd`^js;W>HVp>^_khLmxm>skg65Wduiq=uZ zH`g|x#AN+&U$$}aE9yac3aqV%NMdcwa6}2=2E2NN_@MVly~?82mdboKq%)M?M)KJ{ zyH52jGs-uP<_xBpl10i*bw z_21^z(hD@po;&_vy~wJs7fguq9vsmtTC8RvgB40 zn{}LZye>Z%60j(%b&B;~qGOe_a;3(FeU-3SZ-6K0teM6`hRs$URGeD_7`z@4ZgoQj zaxWaZuy1@$*|?_FIDSvCQ6a4RI9*D`zA^4Dr0#Teq4qfo*1T@(d4Q}v!l5jG!@kA; zP8X5C2ukp^6O!Z-LkAUtxtc4RDjn|@*>HlqJ}{`35ev|Bd$cGQ{L6>AsVQu_MRml6 z3dj3A;(`zRn;Wc-9fi7m3_VrIc7YQAPbxirCHNk#Hh|;ww`8tEAa>{-om_ zegGeg2sDWWKubtz!F@AQd3;8a5ZfZu3ul&9g#-H?-)MJgV^9r|41#G0Hc0hH1I)Ix zcR?ujKQ@t`8*N;0_TI-dou&i#pCWg*TO?iYmZgc)__*^5olc)9i%TpZ6}}HE=+szE0l32U~R8EAUnRW(Qh&R8{IWpH(CAh?g)NN(acg5ESg4L_=UNUe^Og zuwFrVhlv+V1PTo)*!*PlGymECbgXS=blB(9uP2w9As8XU4l%sutjmYUSmE;lHVh zrvzDt{x|2lceR56j!IOV@}HE}!Y_4s*PM)wmG7)S>&M0U&)Q`JlvhwKKFqV{K&ly> z=mVgTxW@r(G-Z%=2^*W$|E0q9Uq7w+NkR|C>}(9 zg;)R19X`XTM1tAT0%6Qh2^twZIt^B}e{ibu zMy*4|?0o0&QJqN^pNZT^WrHDA?4vwqP*y#quAD(o!0A_yug-2C% z>l=CR3^BNthk`kR6|07lFk?f37+yDHN9a#GOH^1A{S7;{XMQzc9NJMof(}|mzy-6Z85MOsGNAFdEfT_>G$u%N zmF>W}!>M3~Ay5;tWqLIP}I{YsFOahaQr2}%LLr9;<$so} z-#^G803HWkw_d5)ByD-=f)i>3L_cCfjPUF?Xj&*Rt(y;nPg5=1U*ShmW=<$f9P#|L z=|$mGZMri{BFl||W_T{}PWAIkJ1|Fv4Eu;MRC&CZhU%9@J$PyK+4i)6`!l<- z^~t;w*w(7V9FFf4$Q2Emx7#V}4G1>`0mYYplb!Xi56PqSwl`jCK4$CdJg>kdLxUZ< zg9<)~X_gWKd?dkv{rQ@i&6qIS`H8pEqH)jh+8eBxAa;M~(70g{WD!%Ir##kJo5V$# z!uIhCR`ovy$@0;On)C>7Bnz{l3wN;`8C|v7wruP71Wajp5uxlmKdNHry`bZ-W=+)= z`rDV0*P9|=0GJ>rWs%&6H4h?tRkPK+OPmJi*#|f4+XAVlc8v5=Xu&-DZ#bB0@Yp!} z267t;kYO8UU_DfYK~yKE=I7n(60Yi&LvsjChHOA*XC@>PCq{nY=(8e7tqx}ZH1dr- z10AyO@Qdg!$uPi;#Qf~WJnd|Rdp!gg3vpr#=Xo(y(=uu_)u*!+Wim@=7g7P5XF%!vV629RH|VrsqFm{oUu13 zVc-%XMWtO|6ywHleOKqNQt$WU$l^$%oZp1aQH&|O#(}Jw_J>-72u2lK)e)xB_hfbL z@Nb#u$g_H*2owpg+v2@3z^nC4uy)NeGtZ-DANX0>I z$4tv)EM%%53Kkg(RwxKC#_9L%Qnx(QzQ`dCB!sizM}dY8)lm)_KSt;Q{<5pU@1dxd zdiV0QAv@b1tS{Y3e2>fv5#fT)eygqQCOg*#YDj&VuOy#gQ|pL}hPSjtP;>}CtJXzb z_*(XE76k(iBI*F{o2jUmWPnTHp8*d>*mIL{Fa`NoiONP!jcS1ZxsQhG^qPoAffiGfcdCL>isW&vx$`ydl5`)Av0x&XvtPW3+zKT%l`P(IW-fMm5a%}(gnXc0ICw@GN2(3(iRPSH==VReYsWMD!4Rv6tF@}buyJZ>~I6nDU z3EtmW5(PRb5l6uS1~0MPpQ0~KjOd`v=B%5@!N;a~F<6Z1Lv2?oPC=M9^y`H)0QGTl zxO|1*-K`0wrYE1CZks8RGpq|X%Go8$wC`)j02(?`iqROTL6VwcUyp zdbHPugc3oS(DpUl;;0dF4H4k31Yms~b`8orm+>p8vgvw3&D0={6<(@8AgsN4fB2dd zef`&PJcw|Co}hP2_+X_{q%*cSa8QBx@hv_s0OJ@cl(dh%G=u2v6SjUXJ$gR6owdYL zk?*M8H(x|4MJpT{?)Ct5B$}ezCy%5E76raDuCYjPDC2d08cxvwelWoWXsE}7bFDRg zPnq0N9665qx-@MxlGC%;7{=COS(HWcUDr?;VXYXw)T)%Uh5^*nngrnL$kZVqDtgB# zSJ-@FZA?#mgKWBaO0aDkg8^$pfBSuHctWQZ8~nzv`vY;j`m-zJf`s-0W+?hmrl2CN z-al*&64>vD-P@WlVli!+g2s9wbUNK4(BmsKcH;z*67kDjjC1%w0~^Viq`M zYOcu~skg>#by5wKxH76Up{J0jPYt=*)8&L)i(g&SM4;z(p{)hAe<)z2$AnE7UPC~9 z)~hIdAjOZ}37c+2v5>hTpB7XN?x!F+wXG=U>(5?|`U4p%6oR6B7LA>4(3CHLn8qU3 z0KT4K_ND-N30k(Oer$M{3`tq&Hx~Nhqvyl~q6yJ$kH!D4eN*flcBhb+{ZSV~mfhRp z)ksSc$7D()&Ht>CB+-fJ*-BkgpzVfGa6NxJZ8%mcDr2rDm?aBJ5swS+42tf_<5XhH}g zXBF6CR35D8P0z0(-5$`$m`f+pH8T605GpL8$xrs2Jez2+2#w+1-(Jio1;efZCbvCm zKZ(8`easNGEi6TtuPAo&9aACxMZZc9GeZWiidTL#m}}W_-c`80-|Q>J$^oL-$3eX5SqSp z@?T81ne&;I;bx{T7VY<-Igp1=rEM!{t_Ourj40)YFjl`!H%wI#@~bxdo-&Xb9~G7q zi4)73r+=p)`luU?eJTo&1u43sZjNYq42nA&8&;G0`qeEF5(48ez`_G@r*^vbkOz8J z+9g5K^H21WPRQ``l$hL&aYU7f0*NvkoS>uOR>I}E5DC%2o4=R@bS%9a^*%!;#KlqB z#_b=btO!uJX$vwih)`Z`3leC>7w(V4SdEfaWU`i)hz*77Jl5N0VuP~Y|3w$8J-t|8 zDnFObad7@RR9ZW{rQiH_p zk4aWg_Dzt_(6mRUH$(0kjaG3Rs&4R_6Wnqszd)CtWdGfRy}!hUGWEjXa?99!;+AyM z6q(#Yv)rW*Pw$HpDE<*on4Ms99sT)4j`h6yp zoT96x<9fb1k zFNQtq2_eZ`8=j@=4}2!IJ4rEYm#jHXIZWLAufOA0A#8jnTDR4b7DZ68y&rSJuwU;q z=&uA0Z&M>kc%IT=9dUNvk_wBa;-}8$$*uEVu!jnU&N1nAYdiy_P*wwK;tHsOWe9R4 zTlAB11j~Z)b{Xt0p`Qy2GTv=M*3-_zodXb$FX~=M>HOtdm#$=SE-f&K2$^#FGe0o? z&yTE4E&PJ(tMXeY;>dTQvgJcu>HFAFmw@-ZuL9fZo-|WeJ`^L-W9*6qYeYrpL<7jM zS8s7CnOSqZ4%MkRREfHRJQFA!K7M0}o(KrPe#SmkVQhmkUGZo9lLQg}S%g4^P#IOr z8LAO?VIbXb{)}L)YfJEa_SN^PvS3d`j}RQ^Vnq1WJQpsTo-43&L}VlTp?HUBZ+i`B za?QKL1L0+RLlAq)QGN~&{7qJ__(Qc=9JxA^UQy6FjfPd)o%XGLr&x0EB;IhqUM zCo7)ZBS$f*vP_>UnN={aLij|~ z(fI?oMsilCCJ_Qr$nae_C{-=M(kA+06p*?E`2fjZ7wDD_LJx78d-!58t z?GQCo-YZHKKcRDJjM=I{)kF77&_RnL4)ltb4QUxz=gAJ-QPVB(;GBNJSCaSSM0Eo$osJd_R)eT;n)iW(PI$4BD)7cqSYqFctkGhJPC0ob5_v_g}01SQ! z=;Gi5ytjG~=LY1NgIdZuoBfh+4dqemFi-;E!8!f8Opdm7007xTgo2Cvr-bs8M9}Z# z+WvXRa`F%L!F!HblNKP60Fi(>gx4_uRVYN`(sYqsmeJreuQPG1H$7plKSNntlCIo^ z!>Zgi@j-E__-|o~2%y6mQaAY~zl5U;o5t`xyK_=#9$S$%fj^gb4(yrQJMhrC{e!S5 z2Y>OEhZ_N0~iqv$o-!SO=Hp2{iY}Pc+5LqRe4f^?=iKXyO{7&TxL>#_!;ey_EajK%G!m6dTG4ODXaz}8N9~w;jKA6?Y1ic?&fib?~2o~ zJN_SE`NzsPiLxR}91VR2fh$cEdmYMa#c4ui|Dsvgzra8leL~wt>RFweYf;MRwt<_+ zcRmRJ;>q&{8QIexHIOg?0Ag7PNnLgkfL62DA-+bGgljuF z`N#f|fPeZ3(a>cR{r%|)lS?v`l@M3-IF(1d4&@C_j-`r&oEU?oI|M9s{{sI2oG5@}(;qeCf(gk3)G54rp11kBB?b z_x$4qgkeYEUqF+;+$LK49imTZW@52FJzgxuRzD|GgqqS=}RBw*nZNJkKgTI=qvP)CTfZh4RN$fxo$=L_JnG`BjHT1UAe`T wCf50rh5dga!E^F)@^SKU@^SL", + "Content-Type": "application/json" + }, + "nonStream": { + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "oauth": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + } + }, + "url": { + "nonStream": "https://logfare.ai/v1/chat/completions", + "stream": "https://logfare.ai/v1/chat/completions" + } + }, "longcat": { "format": "openai", "headers": { diff --git a/tests/unit/logfare-registry.test.ts b/tests/unit/logfare-registry.test.ts new file mode 100644 index 00000000000..d5439cd1b23 --- /dev/null +++ b/tests/unit/logfare-registry.test.ts @@ -0,0 +1,63 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { logfareProvider } from "../../open-sse/config/providers/registry/logfare/index.ts"; + +const { APIKEY_PROVIDERS } = await import( + "../../src/shared/constants/providers.ts" +); +const { REGISTRY: providerRegistry } = + await import("../../open-sse/config/providerRegistry.ts"); +const { NAMED_OPENAI_STYLE_PROVIDERS, isNamedOpenAIStyleProvider } = + await import( + "../../src/app/api/providers/[id]/models/discovery/providerSets.ts" + ); + +const SPEC = { + id: "logfare", + alias: "logfare", + name: "Logfare", + website: "https://logfare.ai", + chatUrl: "https://logfare.ai/v1/chat/completions", + modelsUrl: "https://logfare.ai/v1/models", +}; + +test("logfareProvider registry entry has correct configuration", () => { + assert.equal(logfareProvider.id, "logfare"); + assert.equal(logfareProvider.alias, "logfare"); + assert.equal(logfareProvider.format, "openai"); + assert.equal(logfareProvider.executor, "default"); + assert.equal(logfareProvider.baseUrl, SPEC.chatUrl); + assert.equal(logfareProvider.modelsUrl, SPEC.modelsUrl); + assert.equal(logfareProvider.authType, "apikey"); + assert.equal(logfareProvider.authHeader, "bearer"); + // Catalog is discovered live from /v1/models; no hardcoded seed. + assert.equal(logfareProvider.passthroughModels, true); + assert.equal(logfareProvider.models.length, 0); +}); + +test("APIKEY_PROVIDERS.logfare is registered with the canonical identity", () => { + const entry = APIKEY_PROVIDERS[SPEC.id]; + assert.ok(entry, `APIKEY_PROVIDERS.${SPEC.id} must be defined`); + assert.equal(entry.id, SPEC.id); + assert.equal(entry.alias, SPEC.alias); + assert.equal(entry.name, SPEC.name); + assert.equal(entry.website, SPEC.website); + assert.equal(entry.hasFree, true); + assert.equal(typeof entry.freeNote, "string"); + assert.equal(typeof entry.apiHint, "string"); + assert.match(entry.color, /^#[0-9A-Fa-f]{6}$/); +}); + +test("providerRegistry exposes the OpenAI-compatible chat completions URL", () => { + assert.equal(providerRegistry[SPEC.id].baseUrl, SPEC.chatUrl); + assert.equal(providerRegistry[SPEC.id].modelsUrl, SPEC.modelsUrl); +}); + +test("logfare is classified as a named OpenAI-style provider (live-fetch path)", () => { + assert.ok( + NAMED_OPENAI_STYLE_PROVIDERS.has(SPEC.id), + "logfare must be in NAMED_OPENAI_STYLE_PROVIDERS for live /v1/models fetch" + ); + assert.equal(isNamedOpenAIStyleProvider(SPEC.id), true); +}); diff --git a/tests/unit/providers-constants-split.test.ts b/tests/unit/providers-constants-split.test.ts index 3eabac24bb9..e22b3fd4e83 100644 --- a/tests/unit/providers-constants-split.test.ts +++ b/tests/unit/providers-constants-split.test.ts @@ -25,7 +25,7 @@ // #10729) brings it to 229; Token Kiosk (gateways, #10722) — merged in the same // merge-train batch — independently bumped the gateways family too, landing at 231; Freebuff // (gateways, #10531) brings it to 232. #8864 moves uncloseai (gateways family) into -// NOAUTH_PROVIDERS, dropping the APIKEY_PROVIDERS count to 231. +// NOAUTH_PROVIDERS, dropping the APIKEY_PROVIDERS count to 231. Logfare (gateways, #10987) brings it back to 232. import { test } from "node:test"; import assert from "node:assert/strict"; @@ -54,12 +54,12 @@ test("barrel still exports every catalog + key helpers", () => { } }); -test("APIKEY_PROVIDERS merges the 6 family files into 231 entries (no loss / no dup)", async () => { +test("APIKEY_PROVIDERS merges the 6 family files into 232 entries (no loss / no dup)", async () => { const keys = Object.keys((P as Record).APIKEY_PROVIDERS); - assert.equal(keys.length, 231); - assert.equal(new Set(keys).size, 231, "duplicate keys after spread-merge"); + assert.equal(keys.length, 232); + assert.equal(new Set(keys).size, 232, "duplicate keys after spread-merge"); // the merged object's entry-count equals the sum of the 6 semantic family files; families are a - // strict partition (every provider in exactly one), so the sum must be exactly 231. + // strict partition (every provider in exactly one), so the sum must be exactly 232. const families: [string, string][] = [ ["gateways", "APIKEY_PROVIDERS_GATEWAYS"], ["frontier-labs", "APIKEY_PROVIDERS_FRONTIER"], @@ -79,7 +79,7 @@ test("APIKEY_PROVIDERS merges the 6 family files into 231 entries (no loss / no seen.add(k); } } - assert.equal(famTotal, 231, "families must partition all 231 providers"); + assert.equal(famTotal, 232, "families must partition all 232 providers"); }); test("AI_PROVIDERS Proxy aggregates all sections; lookups resolve", () => { From 79c5bdf681d693add5f96b462d6053f0a6bf151b Mon Sep 17 00:00:00 2001 From: backryun Date: Sun, 23 Aug 2026 11:09:16 +0900 Subject: [PATCH 18/29] fix(release): repair v3.8.50 base-red tail after latest root lift (#10964) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged after conflict triage: the six base-red repair files (vi.json, opencode.ts JSDoc, context-manager test, the three webhook dispatcher tests, the uncloseai orphan-test rename) were already drained on the tip by today's #11130/#11157/#11160/#11113 — those hunks resolved to the tip shape. What lands is the production-fix set: GLM transport-aware Anthropic headers, Claude Code-compatible model-listing rejection, combo live-test single-probe, zero-cost Auto-Combo interval normalization, recovery-clearing union handling, LLMLingua real-path compare, macOS netstat PID discovery, AI Horde R2 strict public-host validation. Sweep of every touched test file: 243/243 green; typecheck + file-size clean. (guide-settings-route's 4 reds reproduce on the pure tip — pre-existing drift from #11079, not from here.) Thank you @backryun! --- docs/reference/ENVIRONMENT.md | 1 + .../providers/registry/uncloseai/index.ts | 1 + open-sse/executors/glm.ts | 19 +++++- .../imageGeneration/providers/aihorde.ts | 5 +- open-sse/services/autoCombo/virtualFactory.ts | 3 +- scripts/build/colocateOptionals.mjs | 6 +- src/app/(dashboard)/dashboard/combos/page.tsx | 2 +- .../[id]/components/HarImportButton.tsx | 2 +- src/app/api/providers/[id]/models/route.ts | 15 +++-- src/i18n/messages/pt-BR.json | 20 +++++- src/lib/services/portProbe.ts | 27 ++++++-- src/sse/handlers/chat.ts | 1 + src/sse/services/auth.ts | 22 +++---- tests/unit/account-rotation-lot-c.test.ts | 31 ++++++--- tests/unit/auth-clear-account-error.test.ts | 5 ++ tests/unit/capture-critical-db-state.test.ts | 2 + tests/unit/cc-compatible-provider.test.ts | 13 ++-- .../unit/cli-helper/config-generator.test.ts | 24 +++---- ...mbo-quota-exhaustion-only-fallback.test.ts | 2 +- .../combo-runtime-unit-concurrency.test.ts | 4 +- tests/unit/cursor-image-input.test.ts | 37 ++++++++++- tests/unit/glm-executor.test.ts | 16 +++-- tests/unit/guide-settings-route.test.ts | 10 ++- ...ard-session-lease-bypass-inventory.test.ts | 2 +- .../unit/mitm-cert-install-mode-9442.test.ts | 7 +- ...outer-free-model-credits-exhausted.test.ts | 14 ++-- .../provider-sweep-live-discovery.test.ts | 12 ++-- tests/unit/readyz-route.test.ts | 4 +- tests/unit/rejected-request-usage.test.ts | 43 +++++++++---- tests/unit/services/ServiceSupervisor.test.ts | 3 + tests/unit/services/portProbePid.test.ts | 6 ++ .../unit/sse-auth-codex-account-pool.test.ts | 4 +- tests/unit/systemd-notify.test.mjs | 12 ++-- tests/unit/terminal-status-origin.test.ts | 64 ++++++++++++++++--- tests/unit/vision-bridge-maxchars.test.ts | 9 +++ 35 files changed, 328 insertions(+), 120 deletions(-) diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index ae06e620905..068c2b02066 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -772,6 +772,7 @@ REQUEST_TIMEOUT_MS (global override) | `KIMI_WEB_BASE_URL` | `https://www.kimi.ai` | Base URL for the Kimi Web (international kimi.ai Connect-RPC) executor (`kimi-web.ts`); override only for mirror/proxy endpoints. | | `KIMI_WEB_CHAT_URL` | `/apiv2/kimi.gateway.chat.v1.ChatService/Chat` | Full chat endpoint for the Kimi Web executor (`kimi-web.ts`). | | `OMNIROUTE_LOGIN_BROWSER_PATH` | _(auto-detected)_ | Path to a system Chrome/Edge executable for the Adobe Firefly interactive browser sign-in (`adobeFireflyBrowserLogin.ts`); overrides per-OS auto-detection. | +| `OMNIROUTE_STANDALONE_DIR` | _.build/ standalone output_ | Build-time override for the standalone output directory consumed by the post-build colocation step (`scripts/build/colocate-standalone.mjs`); build tooling, not runtime. | Combo target attempts inherit the resolved upstream request timeout (`FETCH_TIMEOUT_MS`, or `REQUEST_TIMEOUT_MS` when it supplies the fetch default). Set `targetTimeoutMs` in a combo, diff --git a/open-sse/config/providers/registry/uncloseai/index.ts b/open-sse/config/providers/registry/uncloseai/index.ts index baea064e3c6..2a7b59f5865 100644 --- a/open-sse/config/providers/registry/uncloseai/index.ts +++ b/open-sse/config/providers/registry/uncloseai/index.ts @@ -6,6 +6,7 @@ export const uncloseaiProvider: RegistryEntry = { format: "openai", executor: "default", baseUrl: "https://hermes.ai.unturf.com/v1/chat/completions", + modelsUrl: "https://hermes.ai.unturf.com/v1/models", authType: "optional", authHeader: "bearer", models: [ diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 59e18719076..eda82962557 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -399,7 +399,24 @@ export class GlmExecutor extends DefaultExecutor { ): Promise { const credentials = input.credentials; const url = buildGlmChatUrl(credentials?.providerSpecificData, transport, this.config.baseUrl); - const headers = this.buildHeaders(credentials, input.stream, input.clientHeaders, input.model); + // #10798 moved the transport out of buildHeaders' signature; the Anthropic + // transport must therefore be visible to buildHeaders through + // providerSpecificData (primaryTransport / anthropic-shaped baseUrl). + const headers = + transport === "anthropic" + ? this.buildHeaders( + { + ...credentials, + providerSpecificData: { + ...credentials?.providerSpecificData, + primaryTransport: "anthropic", + }, + }, + input.stream, + input.clientHeaders, + input.model + ) + : this.buildHeaders(credentials, input.stream, input.clientHeaders, input.model); applyConfiguredUserAgent(headers, credentials.providerSpecificData); mergeUpstreamExtraHeaders(headers, input.upstreamExtraHeaders); diff --git a/open-sse/handlers/imageGeneration/providers/aihorde.ts b/open-sse/handlers/imageGeneration/providers/aihorde.ts index 4d39270c034..13fa50f5abf 100644 --- a/open-sse/handlers/imageGeneration/providers/aihorde.ts +++ b/open-sse/handlers/imageGeneration/providers/aihorde.ts @@ -103,11 +103,12 @@ async function fetchHordeImageBytes( if (value.startsWith("http://") || value.startsWith("https://")) { // Horde's response supplies this URL (a signed R2 storage link), not a // fixed OmniRoute-controlled host — route it through the repository's - // established bounded remote-image fetch (SSRF host guard + DNS-rebinding - // pin, streaming byte cap, redirect limit, abort-aware timeout) instead of + // established bounded remote-image fetch (strict public-host validation, + // streaming byte cap, redirect limit, abort-aware timeout) instead of // a bare fetch(). Same helper `imageGeneration.ts` already uses for other // providers' remote image URLs. const remote = await fetchRemoteImage(value, { + guard: "public-only", timeoutMs: options.timeoutMs, signal: options.signal ?? undefined, maxBytes: MAX_HORDE_IMAGE_BYTES, diff --git a/open-sse/services/autoCombo/virtualFactory.ts b/open-sse/services/autoCombo/virtualFactory.ts index 28071144fbb..93d4a0b53e9 100644 --- a/open-sse/services/autoCombo/virtualFactory.ts +++ b/open-sse/services/autoCombo/virtualFactory.ts @@ -7,6 +7,7 @@ import { getProviderRegistry } from "./providerRegistryAccessor"; import type { ConnectionFields } from "@/lib/db/encryption"; import { NOAUTH_PROVIDERS } from "@/shared/constants/providers"; import { hasUsableWebSessionCredential } from "@/shared/providers/webSessionCredentials"; +import { toNumber } from "@/shared/utils/numeric"; import { defaultLogger as log } from "@omniroute/open-sse/utils/logger"; import { getTokenLimit } from "../contextManager"; import { @@ -607,7 +608,7 @@ export async function prepareVirtualAutoComboInputs( // remaining allowance as a percentage, and a raw ">0" comparison would // let a reading of e.g. 0.3% (rounding noise, not real headroom) pass. minRemainingAllowance: 1, - maxStateAgeMs: (Number(settings.autoRefreshProviderQuotaInterval) || 180) * 1000, + maxStateAgeMs: toNumber(settings.autoRefreshProviderQuotaInterval, 180) * 1000, }); if (strictFilteredPool !== pool) pool = strictFilteredPool; diff --git a/scripts/build/colocateOptionals.mjs b/scripts/build/colocateOptionals.mjs index 0aa3f38fab5..7b03a5c53fc 100644 --- a/scripts/build/colocateOptionals.mjs +++ b/scripts/build/colocateOptionals.mjs @@ -47,7 +47,7 @@ * fail-open, so this never throws into the install. */ -import { cpSync, existsSync, mkdirSync, readFileSync } from "node:fs"; +import { cpSync, existsSync, mkdirSync, readFileSync, realpathSync } from "node:fs"; import { createRequire } from "node:module"; import { dirname, join, sep } from "node:path"; @@ -119,7 +119,9 @@ function isPackageIntact(targetNodeModulesDir, name) { const resolved = probe.resolve(name); // A resolution that walked past the target into an ancestor tree does not // prove the target copy is usable. - return resolved.startsWith(targetNodeModulesDir + sep); + const realTarget = realpathSync(targetNodeModulesDir); + const realResolved = realpathSync(resolved); + return realResolved.startsWith(realTarget + sep); } catch { return false; } diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 7ca3af20248..04f73d54a0f 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -533,7 +533,7 @@ function getStrategyBadgeClass(strategy) { return "bg-blue-500/15 text-blue-600 dark:text-blue-400"; } -function getI18nOrFallback(t, key, fallback, values) { +function getI18nOrFallback(t, key, fallback, values = undefined) { try { if (typeof t.has === "function" && t.has(key)) return t(key, values); } catch {} diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/HarImportButton.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/HarImportButton.tsx index 0f4a3d3551c..6066746a459 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/HarImportButton.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/HarImportButton.tsx @@ -68,7 +68,7 @@ export default function HarImportButton({ provider, onImport }: HarImportButtonP } const result = importer(text); - if (!result.ok) { + if (result.ok === false) { const [key, fallback] = ERROR_MESSAGE_KEYS[result.error] ?? [ "harImportErrorUnknown", "Couldn't extract a credential from that HAR file.", diff --git a/src/app/api/providers/[id]/models/route.ts b/src/app/api/providers/[id]/models/route.ts index 01f1bd72857..cecbb6776eb 100755 --- a/src/app/api/providers/[id]/models/route.ts +++ b/src/app/api/providers/[id]/models/route.ts @@ -1909,12 +1909,9 @@ export async function GET( } if (isAnthropicCompatibleProvider(provider)) { - const cachedResponse = maybeReturnCachedDiscovery(); - if (cachedResponse) return cachedResponse; - - const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled(); - if (autoFetchDisabledResponse) return autoFetchDisabledResponse; - + // CC providers never support models listing — this check must precede + // the cached-discovery / auto-fetch fallbacks, which would otherwise + // return a misleading 200 "no models" for a CC node (#10828 ordering). if (isClaudeCodeCompatibleProvider(provider)) { return NextResponse.json( { error: `Provider ${provider} does not support models listing` }, @@ -1922,6 +1919,12 @@ export async function GET( ); } + const cachedResponse = maybeReturnCachedDiscovery(); + if (cachedResponse) return cachedResponse; + + const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled(); + if (autoFetchDisabledResponse) return autoFetchDisabledResponse; + let baseUrl = getProviderBaseUrl(connection.providerSpecificData); if (!baseUrl) { const fallback = buildDiscoveryFallbackResponse({ diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index f3e3c15d130..8df20f0f9df 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -4922,7 +4922,9 @@ "multiProvider": "Multi-Provedor", "usageTracking": "Rastreamento de Uso", "securityDesc": "Defina uma senha para proteger seu painel, ou pule por enquanto.", + "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você poderá adicioná-los depois pelo painel, após definir uma senha.", "providerDesc": "Conecte seu primeiro provedor de IA. Você pode adicionar mais depois.", + "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois pelo painel.", "apiKeyRequired": "Chave de API (obrigatório)", "customUrlOptional": "URL personalizada (opcional)", "testDesc": "Vamos verificar se a conexão com seu provedor funciona.", @@ -4979,9 +4981,7 @@ "skipped": "já configurado", "failed": "falhou" } - }, - "securityDescSkipWarning": "⚠️ Sem uma senha, você não poderá adicionar provedores durante a configuração. Você pode adicioná-los depois no painel após definir uma senha.", - "providerRequiresPassword": "Você precisa definir uma senha primeiro para adicionar provedores. Volte à etapa de segurança e defina uma senha, ou adicione provedores depois no painel." + } }, "providers": { "title": "Provedores", @@ -6269,6 +6269,20 @@ "webSessionGuideStep3": "Copie a credencial necessária do próprio domínio do provedor. Para cookies, copie apenas o valor do cabeçalho Cookie e omita Cookie:.", "webSessionGuideStep3Manual": "Caminho manual: abra as ferramentas do desenvolvedor do navegador (F12 → Network), atualize a página, abra uma requisição autenticada e copie o valor do cabeçalho Cookie em Request Headers — omita o prefixo Cookie:.", "webSessionGuideStep4": "Cole aqui e verifique a conexão. Se parar de funcionar, faça login novamente e substitua-o por um novo valor.", + "harImportButtonLabel": "Importar arquivo .har", + "harImportButtonBusy": "Importando…", + "harImportButtonHint": "Exporte pela aba Rede das Ferramentas do Desenvolvedor após enviar pelo menos uma mensagem no chat.", + "harImportStatusValid": "Importado — válido por cerca de {minutes} min.", + "harImportStatusExpiringSoon": "Importado — válido por apenas mais cerca de {minutes} min.", + "harImportStatusExpired": "Importado, mas este token expirou há {minutes} min — exporte um HAR novo.", + "harImportStatusUnknownExpiry": "Importado. Não foi possível ler a expiração.", + "harImportErrorNotJson": "Esse arquivo não é um JSON válido — ele é realmente uma exportação .har?", + "harImportErrorNoEntries": "Este HAR não contém entradas de rede.", + "harImportErrorNoChathubUrl": "Nenhuma conexão de chat do Copilot foi encontrada neste HAR. Envie pelo menos uma mensagem em m365.cloud.microsoft antes de exportar.", + "harImportErrorUnparsableUrl": "A conexão de chat foi encontrada, mas não foi possível ler a URL.", + "harImportErrorMissingFields": "A conexão de chat foi encontrada, mas o token estava ausente.", + "harImportErrorReadFailed": "Não foi possível ler esse arquivo.", + "harImportErrorUnknown": "Não foi possível extrair uma credencial desse arquivo HAR.", "webSessionSecurityHint": "Trate isso como uma senha: ela poderá acessar sua conta da web conectada até que ela expire ou seja revogada.", "webNoAuthGuideTitle": "Nenhuma credencial necessária", "webNoAuthGuideBody": "{provider} não precisa de chave de API ou cookie. Salve a conexão para usar seu endpoint web gratuito.", diff --git a/src/lib/services/portProbe.ts b/src/lib/services/portProbe.ts index 2a9fd502bda..9a890f541b4 100644 --- a/src/lib/services/portProbe.ts +++ b/src/lib/services/portProbe.ts @@ -168,11 +168,24 @@ export function parseSsPid(stdout: string): number | null { export function parseNetstatPid(stdout: string, port: number): number | null { for (const line of stdout.split("\n")) { const columns = line.trim().split(/\s+/); - // proto recv-q send-q local-address foreign-address state pid/program + // Linux: proto recv-q send-q local-address foreign-address state pid/program if (columns.length < 7 || columns[5] !== "LISTEN") continue; - if (!columns[3].endsWith(`:${port}`)) continue; - const parsed = Number.parseInt(columns[6], 10); - if (Number.isFinite(parsed)) return parsed; + const linuxAddress = columns[3].endsWith(`:${port}`); + const macAddress = columns[3].endsWith(`.${port}`); + if (!linuxAddress && !macAddress) continue; + + if (linuxAddress) { + const linuxPid = Number.parseInt(columns[6], 10); + if (Number.isFinite(linuxPid)) return linuxPid; + } + + // macOS `netstat -anv -p tcp` appends a `process:pid` column after + // the socket counters. Process names may contain spaces, so scan instead + // of relying on one fixed column index. + for (const column of columns.slice(6)) { + const match = /:(\d+)$/.exec(column); + if (match) return Number.parseInt(match[1], 10); + } } return null; } @@ -197,7 +210,11 @@ const PID_PROBES: ReadonlyArray<{ args: (port) => ["-tlnp", `sport = :${port}`], parse: (stdout) => parseSsPid(stdout), }, - { command: "netstat", args: () => ["-tlnp"], parse: parseNetstatPid }, + { + command: "netstat", + args: () => (process.platform === "darwin" ? ["-anv", "-p", "tcp"] : ["-tlnp"]), + parse: parseNetstatPid, + }, ]; /** Run one probe, resolving null on a missing binary, a non-match or a timeout. */ diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 49a29c1b69f..aadf7d29d3d 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -2240,6 +2240,7 @@ async function handleSingleModelChat( if ( !runtimeOptions.emergencyFallbackTried && !comboName && + !forceLiveComboTest && shouldRetrySameAccountTransport({ status: result.status, errorText: errorStr, diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index b1df29552a4..35304a05e69 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -3104,11 +3104,9 @@ export async function clearAccountError( } /** - * Optional CAS token. When provided, the clear is performed via an atomic - * conditional UPDATE (clearConnectionErrorIfUnchanged) that aborts if the row - * was written by a concurrent path between the caller's snapshot read and this - * clear. Closes the TOCTOU window in the quota-recovery path. When omitted, - * the clear is unconditional (preserves existing post-success-call behavior). + * Optional CAS token. When provided, clearConnectionErrorIfUnchanged atomically + * aborts if another path modified the row after the caller's snapshot. + * This closes the TOCTOU window; omission preserves unconditional clearing. */ export interface RecoveredStateExpectation { testStatus: string | null; @@ -3116,23 +3114,25 @@ export interface RecoveredStateExpectation { rateLimitedUntil: string | null; } export async function clearRecoveredProviderState( - credentials: Partial | null, + credentials: unknown, expectedState?: RecoveredStateExpectation ): Promise<{ applied: boolean }> { - if (!credentials?.connectionId) return { applied: false }; + const recoverable = credentials as Partial | null; + if (typeof recoverable?.connectionId !== "string" || !recoverable.connectionId) + return { applied: false }; if (expectedState) { - const applied = await clearConnectionErrorIfUnchanged(credentials.connectionId, expectedState); + const applied = await clearConnectionErrorIfUnchanged(recoverable.connectionId, expectedState); if (!applied) { log.info( "AUTH", - `Skipped recovery clear for ${credentials.connectionId.slice(0, 8)} — state changed concurrently (CAS miss)` + `Skipped recovery clear for ${recoverable.connectionId.slice(0, 8)} — state changed concurrently (CAS miss)` ); return { applied: false }; } - log.info("AUTH", `Account ${credentials.connectionId.slice(0, 8)} error cleared (CAS)`); + log.info("AUTH", `Account ${recoverable.connectionId.slice(0, 8)} error cleared (CAS)`); return { applied: true }; } - await clearAccountError(credentials.connectionId, credentials); + await clearAccountError(recoverable.connectionId, recoverable); return { applied: true }; } type AuthRequestLike = { diff --git a/tests/unit/account-rotation-lot-c.test.ts b/tests/unit/account-rotation-lot-c.test.ts index 0dde306f90a..6ae41c42ed8 100644 --- a/tests/unit/account-rotation-lot-c.test.ts +++ b/tests/unit/account-rotation-lot-c.test.ts @@ -1,6 +1,11 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { pickAccount, markCooldown, markSuccess, isAccountReady } from "../../open-sse/executors/accountRotation.ts"; +import { + pickAccount, + markCooldown, + markSuccess, + isAccountReady, +} from "../../open-sse/executors/accountRotation.ts"; import type { RotatableAccount } from "../../open-sse/executors/accountRotation.ts"; function acct(fp: string, proxy: RotatableAccount["proxy"] = null): RotatableAccount { @@ -11,7 +16,7 @@ test("markCooldown default is transient — no eviction, only backoff", () => { const a = acct("a"); markCooldown(a); // kind omitted → transient assert.ok(a.cooldownUntil > Date.now()); - assert.equal((a as Record).evictedAt, undefined); + assert.equal(a.evictedAt, undefined); // still picked when others are ready const state = { nextAccountIdx: 0 }; const picked = pickAccount([a, acct("b")], state); @@ -25,28 +30,34 @@ test("terminal kind evicts after threshold, pickAccount skips evicted unless all markCooldown(a, "terminal"); markCooldown(a, "terminal"); markCooldown(a, "terminal"); - assert.ok((a as Record).evictedAt != null); + assert.ok(a.evictedAt != null); const state = { nextAccountIdx: 0 }; // b is ready, a evicted → b is picked - const picked = pickAccount([a, b], state, (x) => isAccountReady(x) && !(x as Record).evictedAt); + const picked = pickAccount([a, b], state, (x) => isAccountReady(x) && !x.evictedAt); assert.equal(picked.fingerprint, "healthy"); // when all evicted, caller still gets an account rather than hanging (preserves :52-58) - (b as Record).evictedAt = Date.now(); - const fallback = pickAccount([a, b], { nextAccountIdx: 0 }, (x) => isAccountReady(x) && !(x as Record).evictedAt); + b.evictedAt = Date.now(); + const fallback = pickAccount( + [a, b], + { nextAccountIdx: 0 }, + (x) => isAccountReady(x) && !x.evictedAt + ); assert.ok(fallback.fingerprint === "dead" || fallback.fingerprint === "healthy"); }); test("transient does not evict even after many fails — only terminal does", () => { const a = acct("quota-hit"); for (let i = 0; i < 10; i++) markCooldown(a, "transient"); - assert.equal((a as Record).evictedAt, undefined); + assert.equal(a.evictedAt, undefined); }); test("markSuccess clears eviction and consecutiveFails", () => { const a = acct("revived"); - markCooldown(a, "terminal"); markCooldown(a, "terminal"); markCooldown(a, "terminal"); + markCooldown(a, "terminal"); + markCooldown(a, "terminal"); + markCooldown(a, "terminal"); markSuccess(a); - assert.equal((a as Record).evictedAt, null); + assert.equal(a.evictedAt, null); assert.equal(a.consecutiveFails, 0); }); @@ -57,5 +68,5 @@ test("cross-executor alias still works — opencode wrapper forwards kind", asyn assert.ok(mc.length >= 1 && mc.length <= 2); // Prove it accepts terminal without throw const tmp = acct("probe"); - assert.doesNotThrow(() => (mc as Record)(tmp, "terminal")); + assert.doesNotThrow(() => mc(tmp, "terminal")); }); diff --git a/tests/unit/auth-clear-account-error.test.ts b/tests/unit/auth-clear-account-error.test.ts index 1131ae63701..a303da4e7ed 100644 --- a/tests/unit/auth-clear-account-error.test.ts +++ b/tests/unit/auth-clear-account-error.test.ts @@ -104,6 +104,11 @@ test("clearRecoveredProviderState ignores empty payloads and clears recoverable await auth.clearRecoveredProviderState(null); await auth.clearRecoveredProviderState({}); + await auth.clearRecoveredProviderState({ + allExpired: true, + expiredCount: 1, + expiredStatus: "expired", + }); await auth.clearRecoveredProviderState({ connectionId: created.id, testStatus: "unavailable", diff --git a/tests/unit/capture-critical-db-state.test.ts b/tests/unit/capture-critical-db-state.test.ts index fcf2a017b7e..957c1afce36 100644 --- a/tests/unit/capture-critical-db-state.test.ts +++ b/tests/unit/capture-critical-db-state.test.ts @@ -4,6 +4,8 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +type CoreModule = typeof import("../../src/lib/db/core.ts"); + // Shared across all tests — the module caches DATA_DIR / SQLITE_FILE at load time, // so we must create the temp dir and import exactly once. type CoreModule = typeof import("../../src/lib/db/core.ts"); diff --git a/tests/unit/cc-compatible-provider.test.ts b/tests/unit/cc-compatible-provider.test.ts index 9784382ca38..cbb16a28194 100644 --- a/tests/unit/cc-compatible-provider.test.ts +++ b/tests/unit/cc-compatible-provider.test.ts @@ -1210,13 +1210,8 @@ test("provider models route reports CC compatible providers do not support model { params: { id: connection.id } } ); - assert.ok( - response.status === 400 || response.status === 200, - `CC-compatible models route should 400 (unsupported) or 200 (listed), got ${response.status}` - ); - if (response.status === 400) { - assert.deepEqual(await response.json(), { - error: "Provider anthropic-compatible-cc-test does not support models listing", - }); - } + assert.equal(response.status, 400); + assert.deepEqual(await response.json(), { + error: "Provider anthropic-compatible-cc-test does not support models listing", + }); }); diff --git a/tests/unit/cli-helper/config-generator.test.ts b/tests/unit/cli-helper/config-generator.test.ts index 1b2e85732fe..cfe3f398096 100644 --- a/tests/unit/cli-helper/config-generator.test.ts +++ b/tests/unit/cli-helper/config-generator.test.ts @@ -414,7 +414,7 @@ describe("config-generator", () => { } }); - it("does NOT fabricate a default context when the catalog has no entry", async () => { + it("uses the required 128K context fallback when the catalog has no entry", async () => { const stub = stubFetchOnce(makeCatalogResponse(SAMPLE_CATALOG)); try { const { generateOpencodeConfig } = @@ -424,15 +424,14 @@ describe("config-generator", () => { apiKey: "sk-test", }); const cfg = JSON.parse(out); - // NO_CTX_COMBO has no context_length in the catalog — generator - // must NOT default to 128K (or any other value). The entry is - // emitted without limit.context so OpenCode's own heuristic - // applies and the user can fix the upstream. + // NO_CTX_COMBO has no context_length in the catalog. OpenCode v1 + // requires a complete limit object, so the compatibility fallback + // must be explicit rather than leaving the config invalid. const noCtx = cfg.provider.omniroute.models["NO_CTX_COMBO"]; assert.strictEqual( noCtx.limit?.context, - undefined, - `NO_CTX_COMBO should not have a fabricated limit.context (got ${noCtx.limit?.context})` + 128_000, + `NO_CTX_COMBO should use the 128K fallback (got ${noCtx.limit?.context})` ); } finally { stub.restore(); @@ -603,11 +602,12 @@ describe("config-generator", () => { input: 100000, output: 32768, }); - // #10940: `limit.output` is REQUIRED by OpenCode's v1 provider schema, - // so even a model with zero catalog metadata still gets a `limit` - // block carrying the fallback output value; `context`/`input` stay - // omitted since neither the catalog nor the user knows them. - assert.deepStrictEqual(models["no-metadata"].limit, { output: 8192 }); + // #10940/#11035: OpenCode's v1 provider schema requires both fields, + // so a model with zero metadata gets the compatibility fallbacks. + assert.deepStrictEqual(models["no-metadata"].limit, { + context: 128_000, + output: 8192, + }); for (const model of Object.values(models) as Array<{ limit?: { output?: number } }>) { assert.ok( diff --git a/tests/unit/combo-quota-exhaustion-only-fallback.test.ts b/tests/unit/combo-quota-exhaustion-only-fallback.test.ts index 2a90d7e3fbd..cbb4f70c11f 100644 --- a/tests/unit/combo-quota-exhaustion-only-fallback.test.ts +++ b/tests/unit/combo-quota-exhaustion-only-fallback.test.ts @@ -110,7 +110,7 @@ async function run( } test("quota classifier rejects terminal-looking evidence on ineligible statuses", async () => { - for (const status of [400, 401, 403, 404, 408, 409, 422, 500, 502, 503, 504]) { + for (const status of [400, 401, 404, 408, 409, 422, 500, 502, 503, 504]) { for (const terminal of ["insufficient_quota", "quota_exhausted", "credits_exhausted"]) { assert.equal( await isQuotaExhaustionResponse( diff --git a/tests/unit/combo-runtime-unit-concurrency.test.ts b/tests/unit/combo-runtime-unit-concurrency.test.ts index 36a14a59ce7..8ccebe4603e 100644 --- a/tests/unit/combo-runtime-unit-concurrency.test.ts +++ b/tests/unit/combo-runtime-unit-concurrency.test.ts @@ -28,8 +28,8 @@ const databases = db.pragma("database_list") as Array<{ file?: string; name?: st const activeDbPath = databases.find((database) => database.name === "main")?.file; assert.ok(activeDbPath, "test requires a file-backed main SQLite database"); assert.equal( - path.dirname(path.resolve(activeDbPath)), - path.resolve(TEST_DATA_DIR), + fs.realpathSync(path.dirname(path.resolve(activeDbPath))), + fs.realpathSync(path.resolve(TEST_DATA_DIR)), `active test database must be under TEST_DATA_DIR before inserts: ${activeDbPath}` ); diff --git a/tests/unit/cursor-image-input.test.ts b/tests/unit/cursor-image-input.test.ts index 16bffcb6b30..abc14d7d983 100644 --- a/tests/unit/cursor-image-input.test.ts +++ b/tests/unit/cursor-image-input.test.ts @@ -1,4 +1,4 @@ -import test from "node:test"; +import test, { type TestContext } from "node:test"; import assert from "node:assert/strict"; import crypto from "node:crypto"; import dns from "node:dns"; @@ -473,7 +473,37 @@ test("resolveCursorImages soft-caps a large PNG under the wire budget", async () // ─── Executor-level error body (response path, hard rule #12) ─────────────── -test("executor returns a sanitized 400 for an oversized image", async () => { +// #10804 moved agent-endpoint discovery (a live api2.cursor.sh call) ahead of +// request building inside CursorExecutor.execute. These tests exercise the +// image-validation 400 path with a fake token, so stub the discovery fetch to +// return a minimal valid Connect-RPC config response instead of hitting the +// network (which would 401 before image validation ever runs). +function mockCursorServerConfig(t: TestContext): void { + t.mock.method(globalThis, "fetch", async (input, init) => { + const url = String(input); + if (!url.includes("ServerConfigService/GetServerConfig")) { + throw new Error(`unexpected fetch in test: ${url}`); + } + void init; + // Minimal protobuf matching parseCursorAgentUrls: field 27 wraps a + // sub-message holding field 1 (agentUrl) + field 2 (agentnUrl), each a + // length-delimited https://host string. validateCursorAgentUrl only + // accepts *.api5.cursor.sh hosts, so use those. + const str = (field: number, host: string): Buffer => { + const value = Buffer.from(`https://${host}`); + return Buffer.concat([Buffer.from([(field << 3) | 0x02, value.length]), value]); + }; + const inner = Buffer.concat([str(1, "us.api5.cursor.sh"), str(2, "eu.api5.cursor.sh")]); + // Field-27 tag (218) needs proper varint encoding (2 bytes). + const tag = ((27 << 3) | 0x02) as number; + const header = Buffer.from([(tag & 0x7f) | 0x80, tag >>> 7, inner.length]); + const body = Buffer.concat([header, inner]); + return new Response(body, { status: 200 }); + }); +} + +test("executor returns a sanitized 400 for an oversized image", async (t) => { + mockCursorServerConfig(t); const exec = new CursorExecutor(); const big = Buffer.alloc(MAX_CURSOR_IMAGE_DECODE_BYTES + 16).toString("base64"); const result = await exec.execute({ @@ -508,7 +538,8 @@ test("executor returns a sanitized 400 for an oversized image", async () => { assert.ok(!/\/(root|home|usr)\//.test(body.error.message), "no absolute path in error body"); }); -test("executor returns a sanitized 400 for an SSRF-blocked image URL", async () => { +test("executor returns a sanitized 400 for an SSRF-blocked image URL", async (t) => { + mockCursorServerConfig(t); const exec = new CursorExecutor(); const result = await exec.execute({ model: "gpt-5.2", diff --git a/tests/unit/glm-executor.test.ts b/tests/unit/glm-executor.test.ts index 4c1a4942352..3d47b375c35 100644 --- a/tests/unit/glm-executor.test.ts +++ b/tests/unit/glm-executor.test.ts @@ -164,7 +164,12 @@ test("GlmExecutor separates OpenAI-compatible coding headers from Anthropic head const anthropicHeaders = executor.buildHeaders( { apiKey: "glm-key", - providerSpecificData: { baseUrl: "https://api.z.ai/api/anthropic/v1/messages" }, + providerSpecificData: { + baseUrl: "https://api.z.ai/api/anthropic/v1/messages", + // Same #10798 signature change — Anthropic transport via + // providerSpecificData (baseUrl is anthropic-shaped anyway). + primaryTransport: "anthropic", + }, }, true, null, @@ -191,6 +196,8 @@ test("GlmExecutor preserves extra API key rotation", () => { connectionId: "glm-rotation-test", providerSpecificData: { baseUrl: "https://api.z.ai/api/anthropic/v1/messages", + // #10798 signature change — Anthropic transport via providerSpecificData. + primaryTransport: "anthropic", extraApiKeys: ["extra-key"], }, }, @@ -426,10 +433,9 @@ test("GlmExecutor falls back internally to Anthropic transport and returns OpenA assert.equal(calls[0].url, "https://api.z.ai/api/coding/paas/v4/chat/completions"); assert.equal(calls[0].headers.Authorization, "Bearer glm-key"); assert.equal(calls[1].url, "https://api.z.ai/api/anthropic/v1/messages?beta=true"); - const fallbackKey = - calls[1].headers["x-api-key"] || - String(calls[1].headers.Authorization || "").replace(/^Bearer\s+/i, ""); - assert.equal(fallbackKey, "glm-key"); + assert.equal(calls[1].headers["x-api-key"], "glm-key"); + assert.equal(calls[1].headers.Authorization, undefined); + assert.equal(calls[1].headers["anthropic-version"], "2023-06-01"); assert.equal(calls[1].body.messages[0].role, "user"); assert.equal(calls[1].body._disableToolPrefix, undefined); assert.equal(result.targetFormat, "openai"); diff --git a/tests/unit/guide-settings-route.test.ts b/tests/unit/guide-settings-route.test.ts index 735240dc1e9..7516bd0be5c 100644 --- a/tests/unit/guide-settings-route.test.ts +++ b/tests/unit/guide-settings-route.test.ts @@ -198,8 +198,14 @@ test("guide-settings POST preserves existing OpenCode config fields while only u assert.equal(content.provider.omniroute.options.baseURL, "http://my-omni/v1"); assert.ok(content.provider.omniroute.options.apiKey.startsWith("sk-")); assert.deepEqual(content.provider.omniroute.models, { - "cx/gpt-5.6-sol": { name: "GPT-5.6 Sol" }, - "opencode-go/kimi-k2.6": { name: "Kimi K2.6" }, + "cx/gpt-5.6-sol": { + name: "GPT-5.6 Sol", + limit: { context: 128_000, output: 8192 }, + }, + "opencode-go/kimi-k2.6": { + name: "Kimi K2.6", + limit: { context: 128_000, output: 8192 }, + }, }); }); diff --git a/tests/unit/hard-session-lease-bypass-inventory.test.ts b/tests/unit/hard-session-lease-bypass-inventory.test.ts index bafa2971460..396c54015c7 100644 --- a/tests/unit/hard-session-lease-bypass-inventory.test.ts +++ b/tests/unit/hard-session-lease-bypass-inventory.test.ts @@ -86,7 +86,7 @@ const EXPECTED: Record> = { "src/app/api/providers/client/route.ts": 1, "src/app/api/providers/free-onboarding/route.ts": 2, "src/app/api/providers/import/route.ts": 1, - "src/app/api/providers/route.ts": 4, + "src/app/api/providers/route.ts": 2, "src/app/api/providers/test-batch/route.ts": 2, "src/app/api/rate-limits/route.ts": 1, "src/app/api/services/dario/admin/import-from-omniroute/route.ts": 2, diff --git a/tests/unit/mitm-cert-install-mode-9442.test.ts b/tests/unit/mitm-cert-install-mode-9442.test.ts index 979dee7d4c8..4128719c5cd 100644 --- a/tests/unit/mitm-cert-install-mode-9442.test.ts +++ b/tests/unit/mitm-cert-install-mode-9442.test.ts @@ -165,11 +165,14 @@ test("filesystem proof: cp under umask 0077 creates mode 0600 (why the fix is ne const oldUmask = process.umask(0o077); try { - // Use the real `cp` (GNU coreutils) by absolute path — the exact command + // Use the real `cp` (GNU/BSD coreutils) by absolute path — the exact command // installCertLinux runs — so the umask actually applies. Node's // fs.copyFileSync preserves the source mode, which would mask the bug, and // the bare `cp` on PATH below is a logging stub from the install tests. - execFileSync("/usr/bin/cp", [src, dst]); + // macOS keeps coreutils at /bin/cp; Linux (GNU coreutils) at /usr/bin/cp. + const realCp = ["/usr/bin/cp", "/bin/cp"].find((p) => fs.existsSync(p)); + assert.ok(realCp, "a real cp binary must exist for this filesystem proof"); + execFileSync(realCp, [src, dst]); const mode = fs.statSync(dst).mode & 0o777; assert.equal(mode, 0o600, "cp under umask 0077 must produce 0600 — the bug this fix repairs"); } finally { diff --git a/tests/unit/openrouter-free-model-credits-exhausted.test.ts b/tests/unit/openrouter-free-model-credits-exhausted.test.ts index 601553b658b..129d2fac6d9 100644 --- a/tests/unit/openrouter-free-model-credits-exhausted.test.ts +++ b/tests/unit/openrouter-free-model-credits-exhausted.test.ts @@ -78,7 +78,11 @@ test("getProviderCredentials still refuses a PAID OpenRouter model on a credits_ "anthropic/claude-opus-4.5" ); - assert.equal(selected, null, "paid-model requests must still be blocked on the exhausted connection"); + assert.deepEqual( + selected, + { allExpired: true, expiredCount: 1, expiredStatus: "credits_exhausted" }, + "paid-model requests must still be blocked on the exhausted connection" + ); }); test("getProviderCredentials still refuses a :free OpenRouter model on a banned connection", async () => { @@ -99,9 +103,9 @@ test("getProviderCredentials still refuses a :free OpenRouter model on a banned "meta-llama/llama-3.1-8b-instruct:free" ); - assert.equal( + assert.deepEqual( selected, - null, + { allExpired: true, expiredCount: 1, expiredStatus: "banned" }, "the free-model exemption only applies to credits_exhausted, not other terminal statuses" ); }); @@ -119,9 +123,9 @@ test("getProviderCredentials still refuses a :free model on a credits_exhausted const selected = await auth.getProviderCredentials("openai", null, null, "some-model:free"); - assert.equal( + assert.deepEqual( selected, - null, + { allExpired: true, expiredCount: 1, expiredStatus: "credits_exhausted" }, "the exemption is OpenRouter-specific, since only OpenRouter uses the :free naming convention with a shared balance" ); }); diff --git a/tests/unit/provider-sweep-live-discovery.test.ts b/tests/unit/provider-sweep-live-discovery.test.ts index 323a1599f3e..2431179fa7a 100644 --- a/tests/unit/provider-sweep-live-discovery.test.ts +++ b/tests/unit/provider-sweep-live-discovery.test.ts @@ -48,7 +48,7 @@ interface ModelsBody { } // provider → the upstream /models URL the route resolves from its registry baseUrl. -const LIVE_CASES: Array<{ provider: string; liveUrl: string }> = [ +const LIVE_CASES: Array<{ provider: string; liveUrl: string; source?: string }> = [ { provider: "venice", liveUrl: "https://api.venice.ai/api/v1/models" }, { provider: "deepinfra", liveUrl: "https://api.deepinfra.com/v1/openai/models" }, { provider: "wandb", liveUrl: "https://api.inference.wandb.ai/v1/models" }, @@ -62,7 +62,11 @@ const LIVE_CASES: Array<{ provider: string; liveUrl: string }> = [ { provider: "ovhcloud", liveUrl: "https://oai.endpoints.kepler.ai.cloud.ovh.net/v1/models" }, { provider: "sambanova", liveUrl: "https://api.sambanova.ai/v1/models" }, { provider: "orcarouter", liveUrl: "https://api.orcarouter.ai/v1/models" }, - { provider: "uncloseai", liveUrl: "https://hermes.ai.unturf.com/v1/models" }, + { + provider: "uncloseai", + liveUrl: "https://hermes.ai.unturf.com/v1/models", + source: "upstream", + }, { provider: "opencode-go", liveUrl: "https://opencode.ai/zen/go/v1/models" }, { provider: "baseten", liveUrl: "https://inference.baseten.co/v1/models" }, { provider: "hyperbolic", liveUrl: "https://api.hyperbolic.xyz/v1/models" }, @@ -76,7 +80,7 @@ const LIVE_CASES: Array<{ provider: string; liveUrl: string }> = [ { provider: "api-airforce", liveUrl: "https://api.airforce/v1/models" }, ]; -for (const { provider, liveUrl } of LIVE_CASES) { +for (const { provider, liveUrl, source = "api" } of LIVE_CASES) { test(`sweep: ${provider} import fetches the live /models catalog`, async () => { await resetStorage(); const connection = await providersDb.createProviderConnection({ @@ -109,7 +113,7 @@ for (const { provider, liveUrl } of LIVE_CASES) { const body = (await response.json()) as ModelsBody; assert.equal(body.provider, provider); assert.ok(fetched, `should have probed ${liveUrl}`); - assert.equal(body.source, "api", "should serve the live upstream catalog, not local_catalog"); + assert.equal(body.source, source, "should serve the live upstream catalog, not local_catalog"); const ids = body.models.map((m) => m.id); assert.ok( ids.includes(`${provider}-live-a`) && ids.includes(`${provider}-live-b`), diff --git a/tests/unit/readyz-route.test.ts b/tests/unit/readyz-route.test.ts index ec9ed1cb12b..ef5c0f67ee5 100644 --- a/tests/unit/readyz-route.test.ts +++ b/tests/unit/readyz-route.test.ts @@ -49,9 +49,11 @@ test("/readyz is omitted from the centralized auth proxy matcher", () => { assert.equal(/["']\/healthz/.test(matcherBlock), false); }); -test("/readyz re-exports the /healthz handlers (no second lifecycle)", () => { +test("/readyz re-exports the /healthz handlers and declares its route config locally", () => { const source = fs.readFileSync("src/app/readyz/route.ts", "utf8"); assert.match(source, /from ["']\.\.\/healthz\/route["']/); + assert.match(source, /export const dynamic = ["']force-dynamic["']/); + assert.doesNotMatch(source, /export\s*\{[^}]*\bdynamic\b[^}]*\}\s*from/); assert.equal(/monitoring/i.test(source), false); assert.equal(/sqlite/i.test(source), false); }); diff --git a/tests/unit/rejected-request-usage.test.ts b/tests/unit/rejected-request-usage.test.ts index a25ada202b2..300e77e013c 100644 --- a/tests/unit/rejected-request-usage.test.ts +++ b/tests/unit/rejected-request-usage.test.ts @@ -60,12 +60,17 @@ test("gate-rejected request is attributed to the api key in usage_history", asyn assert.equal(keyRows.length, 1, "expected one usage_history row for the rejected request"); assert.equal(keyRows[0].success, false, "rejected request must be recorded as success:false"); - // call_logs visibility is preserved (dashboard/logs). - const logs = await callLogs.getCallLogs({}); - const rejected = (logs.logs ?? logs).filter?.( - (l: { apiKeyName?: string | null }) => l.apiKeyName === "opencode-mac" - ); - assert.ok(rejected && rejected.length >= 1, "expected a call_logs row for the rejected request"); + // call_logs visibility is preserved (dashboard/logs). saveCallLog is + // fire-and-forget inside recordRejectedRequestUsage, so poll briefly for the + // row instead of asserting synchronously after the await. + let rejected: Array<{ apiKeyName?: string | null }> = []; + for (let i = 0; i < 50 && rejected.length === 0; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + rejected = (list ?? []).filter((l) => l.apiKeyName === "opencode-mac"); + if (rejected.length === 0) await new Promise((r) => setTimeout(r, 10)); + } + assert.ok(rejected.length >= 1, "expected a call_logs row for the rejected request"); }); test("combo-exhausted rejection is also counted per api key", async () => { @@ -111,10 +116,15 @@ test("combo-exhausted rejection persists the client request body for dashboard i requestBody: { model: "default", messages: [{ role: "user", content: "hello" }] }, }); - const logs = await callLogs.getCallLogs({}); - const rejected = (logs.logs ?? logs).find?.( - (l: { apiKeyName?: string | null }) => l.apiKeyName === "request-body-test" - ); + // saveCallLog is fire-and-forget — poll briefly for the row. + let rejected: { id: string; hasRequestBody: boolean } | undefined; + for (let i = 0; i < 50 && !rejected; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + const found = (list ?? []).find((l) => l.apiKeyName === "request-body-test"); + if (found) rejected = found as unknown as { id: string; hasRequestBody: boolean }; + else await new Promise((r) => setTimeout(r, 10)); + } assert.ok(rejected, "expected a call_logs row for the rejected request"); assert.equal(rejected.hasRequestBody, true, "expected hasRequestBody to be true"); @@ -140,10 +150,15 @@ test("combo-exhausted rejection without a request body still logs cleanly (no re startTime: Date.now() - 100, }); - const logs = await callLogs.getCallLogs({}); - const rejected = (logs.logs ?? logs).find?.( - (l: { apiKeyName?: string | null }) => l.apiKeyName === "no-body-test" - ); + // saveCallLog is fire-and-forget — poll briefly for the row. + let rejected: { id: string } | undefined; + for (let i = 0; i < 50 && !rejected; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + const found = (list ?? []).find((l) => l.apiKeyName === "no-body-test"); + if (found) rejected = found as unknown as { id: string }; + else await new Promise((r) => setTimeout(r, 10)); + } assert.ok(rejected, "expected a call_logs row even without a request body"); assert.equal(rejected.hasRequestBody, false); }); diff --git a/tests/unit/services/ServiceSupervisor.test.ts b/tests/unit/services/ServiceSupervisor.test.ts index 011395733ad..6e04455e614 100644 --- a/tests/unit/services/ServiceSupervisor.test.ts +++ b/tests/unit/services/ServiceSupervisor.test.ts @@ -18,6 +18,9 @@ const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-superviso process.env.DATA_DIR = TEST_DATA_DIR; process.env.NODE_ENV = "test"; process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; +// Adoption is intentionally opt-in after GHSA-wg9p-6m2g-4v27. These tests +// exercise the explicit adoption path, so enable it for this isolated process. +process.env.OMNIROUTE_ADOPT_EXISTING_SERVICE = "1"; // Import DB core first to trigger migration (creates version_manager with new columns) const core = await import("../../../src/lib/db/core.ts"); diff --git a/tests/unit/services/portProbePid.test.ts b/tests/unit/services/portProbePid.test.ts index b88383fd801..c58bf86045c 100644 --- a/tests/unit/services/portProbePid.test.ts +++ b/tests/unit/services/portProbePid.test.ts @@ -64,6 +64,12 @@ test("parseNetstatPid matches on the local address, not the foreign one", () => assert.equal(parseNetstatPid(stdout, 20128), 596922); }); +test("parseNetstatPid reads macOS process:pid output", () => { + const stdout = + "tcp4 0 0 127.0.0.1.20128 *.* LISTEN 0 0 131072 131072 node:596922 00100\n"; + assert.equal(parseNetstatPid(stdout, 20128), 596922); +}); + test("parseNetstatPid ignores non-listening rows and unknown ports", () => { const stdout = "tcp 0 0 127.0.0.1:20128 1.2.3.4:5555 ESTABLISHED 596922/node\n"; diff --git a/tests/unit/sse-auth-codex-account-pool.test.ts b/tests/unit/sse-auth-codex-account-pool.test.ts index 581efd1f339..927775d15f7 100644 --- a/tests/unit/sse-auth-codex-account-pool.test.ts +++ b/tests/unit/sse-auth-codex-account-pool.test.ts @@ -242,8 +242,8 @@ test("Codex parent authentication failures block both virtual children without c const inventory = await providersDb.getProviderConnections({ provider: "codex" }); assert.equal(unavailable.shouldFallback, true); - assert.equal(spark, null); - assert.equal(normal, null); + assert.deepEqual(spark, { allExpired: true, expiredCount: 1, expiredStatus: "expired" }); + assert.deepEqual(normal, { allExpired: true, expiredCount: 1, expiredStatus: "expired" }); assert.deepEqual( inventory.map((item) => item.id), [connection.id] diff --git a/tests/unit/systemd-notify.test.mjs b/tests/unit/systemd-notify.test.mjs index cb9e441f8f7..b716d57d9d5 100644 --- a/tests/unit/systemd-notify.test.mjs +++ b/tests/unit/systemd-notify.test.mjs @@ -37,8 +37,10 @@ function python3Available() { } } -// Waits for the listener to emit `expected` lines (in order), then resolves -// with everything it saw. Fails loudly on timeout or premature exit. +// Waits for the listener to emit every `expected` line, then resolves with +// everything it saw. The notifier spawns one process per signal, so AF_UNIX +// datagram arrival order is not guaranteed across those processes. +// Fails loudly on timeout or premature exit. // BARRIER=1 datagrams (sd_notify synchronization emitted by the systemd-notify // CLI after every message) are noise for this contract and are skipped. function waitForLines(child, expected, timeoutMs) { @@ -58,14 +60,14 @@ function waitForLines(child, expected, timeoutMs) { buf = buf.slice(idx + 1); if (!line || line === "BARRIER=1") continue; seen.push(line); - if (seen.length === expected.length) { + if (expected.every((expectedLine) => seen.includes(expectedLine))) { clearTimeout(timer); resolve([...seen]); } } }); child.on("exit", () => { - if (seen.length < expected.length) { + if (expected.some((expectedLine) => !seen.includes(expectedLine))) { clearTimeout(timer); reject(new Error(`listener exited early; got: ${seen.join(", ")}`)); } @@ -248,7 +250,7 @@ test( notifier.watchdog(); notifier.stopping(); const received = await waitForLines(listener, ["READY=1", "WATCHDOG=1", "STOPPING=1"], 10000); - assert.deepEqual(received, ["READY=1", "WATCHDOG=1", "STOPPING=1"]); + assert.deepEqual(received.toSorted(), ["READY=1", "STOPPING=1", "WATCHDOG=1"]); notifier.dispose(); } finally { listener.kill(); diff --git a/tests/unit/terminal-status-origin.test.ts b/tests/unit/terminal-status-origin.test.ts index a4aacf767fc..3a9221f2cab 100644 --- a/tests/unit/terminal-status-origin.test.ts +++ b/tests/unit/terminal-status-origin.test.ts @@ -1,6 +1,8 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; const DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-t11-")); process.env.DATA_DIR = DIR; @@ -9,15 +11,42 @@ const { createProviderConnection } = await import("../../src/lib/db/providers.ts const { runAsProbe } = await import("../../src/shared/utils/probeOrigin.ts"); const { writeTerminalStatus } = await import("../../src/shared/utils/terminalStatus.ts"); -test.after(() => { core.resetDbInstance(); fs.rmSync(DIR, {recursive:true, force:true}); }); +test.after(() => { + core.resetDbInstance(); + fs.rmSync(DIR, { recursive: true, force: true }); +}); -function row(id: string){ return (core.getDbInstance() as unknown as Record).prepare("SELECT is_active, test_status FROM provider_connections WHERE id=?").get(id); } +function row(id: string): { is_active: number; test_status: string } { + const result = core + .getDbInstance() + .prepare("SELECT is_active, test_status FROM provider_connections WHERE id=?") + .get(id); + assert.ok(result && typeof result === "object"); + return result as { is_active: number; test_status: string }; +} test("probe-origin writeTerminalStatus records error but never deactivates", async () => { - const conn = await createProviderConnection({ provider:"openai", authType:"apikey", name:"t11", apiKey:"sk-t11", isActive:true, testStatus:"active" } as unknown as Record); - const id = String((conn as unknown as Record).id); + const conn = await createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "t11", + apiKey: "sk-t11", + isActive: true, + testStatus: "active", + }); + const id = String(conn.id); await runAsProbe(async () => { - await writeTerminalStatus(id, { testStatus:"banned", isActive:false, lastError:"probe 403", errorCode:"403", lastErrorType:"FORBIDDEN" }, "probe"); + await writeTerminalStatus( + id, + { + testStatus: "banned", + isActive: false, + lastError: "probe 403", + errorCode: "403", + lastErrorType: "FORBIDDEN", + }, + "probe" + ); }); const r = row(id); assert.equal(r.is_active, 1); // probe n'a jamais désactivé @@ -25,9 +54,26 @@ test("probe-origin writeTerminalStatus records error but never deactivates", asy }); test("production writeTerminalStatus deactivates on terminal", async () => { - const conn = await createProviderConnection({ provider:"openai", authType:"apikey", name:"t11b", apiKey:"sk-t11b", isActive:true, testStatus:"active" } as unknown as Record); - const id = String((conn as unknown as Record).id); - await writeTerminalStatus(id, { testStatus:"banned", isActive:false, lastError:"real 403", errorCode:"403", lastErrorType:"FORBIDDEN" }, "production"); + const conn = await createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "t11b", + apiKey: "sk-t11b", + isActive: true, + testStatus: "active", + }); + const id = String(conn.id); + await writeTerminalStatus( + id, + { + testStatus: "banned", + isActive: false, + lastError: "real 403", + errorCode: "403", + lastErrorType: "FORBIDDEN", + }, + "production" + ); const r = row(id); assert.equal(r.is_active, 0); assert.equal(r.test_status, "banned"); diff --git a/tests/unit/vision-bridge-maxchars.test.ts b/tests/unit/vision-bridge-maxchars.test.ts index d667e515621..7f364468c73 100644 --- a/tests/unit/vision-bridge-maxchars.test.ts +++ b/tests/unit/vision-bridge-maxchars.test.ts @@ -67,6 +67,11 @@ test("modalityBridgeVisionMaxChars=120 caps the description with a … suffix", }), callVisionModel: async (_imageDataUri: string, _config: VisionModelConfig) => LONG_DESCRIPTION, + // #10859 made the reroute heuristic try a live vision-capable model for + // not-combo text-only models, which would hijack the request before the + // describe path. Pin credentials to definitively-unusable (false) so the + // reroute is excluded and the describe path under test runs. + hasUsableCredentials: async () => false, }, }); @@ -93,6 +98,8 @@ test("no modalityBridgeVisionMaxChars key: description is passed through in full }), callVisionModel: async (_imageDataUri: string, _config: VisionModelConfig) => LONG_DESCRIPTION, + // Same #10859 reroute guard as above — keep the describe path under test. + hasUsableCredentials: async () => false, }, }); @@ -125,6 +132,8 @@ test("updateSettingsSchema accepts an explicit modalityBridgeVisionMaxChars: 0 t }), callVisionModel: async (_imageDataUri: string, _config: VisionModelConfig) => LONG_DESCRIPTION, + // Same #10859 reroute guard as above — keep the describe path under test. + hasUsableCredentials: async () => false, }, }); From f131b64a6e2130dafe2505deab1fb8ff7d6cdbea Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Rouzbeh=E2=80=A0?= <78313022+rqzbeh@users.noreply.github.com> Date: Sun, 23 Aug 2026 07:30:14 +0330 Subject: [PATCH 19/29] fix(security): add test coverage for Tier 1 local-only route guard process-spawning endpoints (#11189) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board (gates + typecheck clean) and this branch: security-route-guard-tiers green. Regression coverage for the Hard Rule #15/#17 contract — Tier 1 process-spawning prefixes (/api/services/, /api/mcp/, /api/cli-tools/runtime/) must stay LOCAL_ONLY before any auth check. Conflict with the tip was only stale provider-count docs. Thank you @rqzbeh! --- config/quality/eslint-suppressions.json | 12 +----------- tests/unit/security-route-guard-tiers.test.ts | 10 ++++++++++ 2 files changed, 11 insertions(+), 11 deletions(-) create mode 100644 tests/unit/security-route-guard-tiers.test.ts diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index 3dae8591dd7..79875e31489 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -853,11 +853,6 @@ "count": 1 } }, - "src/app/api/usage/call-logs/route.ts": { - "no-restricted-imports": { - "count": 1 - } - }, "src/app/api/usage/quota/route.ts": { "no-restricted-imports": { "count": 1 @@ -953,11 +948,6 @@ "count": 1 } }, - "src/app/api/v1/rerank/route.ts": { - "no-restricted-imports": { - "count": 1 - } - }, "src/app/api/v1/vscode/[token]/models/route.ts": { "no-restricted-syntax": { "count": 1 @@ -3259,4 +3249,4 @@ "count": 5 } } -} +} \ No newline at end of file diff --git a/tests/unit/security-route-guard-tiers.test.ts b/tests/unit/security-route-guard-tiers.test.ts new file mode 100644 index 00000000000..dfc1f3c026f --- /dev/null +++ b/tests/unit/security-route-guard-tiers.test.ts @@ -0,0 +1,10 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { isLocalOnlyPath } from "../../src/server/authz/routeGuard.ts"; + +test("isLocalOnlyPath correctly classifies process-spawning endpoints under Tier 1 LOCAL_ONLY", () => { + assert.equal(isLocalOnlyPath("/api/services/dario/start"), true); + assert.equal(isLocalOnlyPath("/api/mcp/stream"), true); + assert.equal(isLocalOnlyPath("/api/cli-tools/runtime/status"), true); + assert.equal(isLocalOnlyPath("/api/v1/chat/completions"), false); +}); From b25b2eacb3bd7c4a82f1b1c12512cff212d28323 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Rouzbeh=E2=80=A0?= <78313022+rqzbeh@users.noreply.github.com> Date: Sun, 23 Aug 2026 07:31:11 +0330 Subject: [PATCH 20/29] refactor(dashboard): format custom provider quota keys into title-cased labels (#11188) Validated on the combined batch board + this branch: dashboard-ux-operability green. Unmapped custom quota keys now render as title-cased labels instead of raw snake_case. Conflict with the tip was only stale provider-count docs. Thank you @rqzbeh! --- .../dashboard/usage/components/ProviderLimits/utils.tsx | 2 +- tests/unit/dashboard-ux-operability.test.ts | 9 +++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) create mode 100644 tests/unit/dashboard-ux-operability.test.ts diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx index 3af32da285d..dc102936d09 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx @@ -98,7 +98,7 @@ export function formatQuotaLabel(name: string) { return `Weekly ${toTitleCaseWords(weeklyModelMatch[1])}`; } - return trimmed; + return toTitleCaseWords(trimmed.replace(/_/g, " ")); } /** diff --git a/tests/unit/dashboard-ux-operability.test.ts b/tests/unit/dashboard-ux-operability.test.ts new file mode 100644 index 00000000000..cd3e3b533ad --- /dev/null +++ b/tests/unit/dashboard-ux-operability.test.ts @@ -0,0 +1,9 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { formatQuotaLabel } from "../../src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx"; + +test("formatQuotaLabel formats custom quota keys with proper title-casing", () => { + assert.equal(formatQuotaLabel("session"), "Session"); + assert.equal(formatQuotaLabel("weekly"), "Weekly"); + assert.equal(formatQuotaLabel("custom_quota_limit"), "Custom Quota Limit"); +}); From 47147e0bcd5aa1d4f42df04c5b4e9e9a1f792ac4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Rouzbeh=E2=80=A0?= <78313022+rqzbeh@users.noreply.github.com> Date: Sun, 23 Aug 2026 07:33:40 +0330 Subject: [PATCH 21/29] fix(auth): replace router.push with window.location navigation after login (#11143) (#11175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board + this branch: login-11143 green. Full document navigation after login guarantees the auth_token cookie is committed before any RSC prefetch fires — no more 307 back to /login. Conflict with the tip was only stale provider-count docs. Fixes #11143. Thank you @rqzbeh! --- src/app/login/page.tsx | 8 +++----- tests/unit/login-11143.test.ts | 21 +++++++++++++++++++++ 2 files changed, 24 insertions(+), 5 deletions(-) create mode 100644 tests/unit/login-11143.test.ts diff --git a/src/app/login/page.tsx b/src/app/login/page.tsx index f2c70306f38..664eb3c1721 100644 --- a/src/app/login/page.tsx +++ b/src/app/login/page.tsx @@ -38,8 +38,7 @@ export default function LoginPage() { if (data.nodeVersion) setNodeVersion(data.nodeVersion); if (data.nodeCompatible === false) setNodeCompatible(false); if (data.authenticated === true || data.requireLogin === false) { - router.push("/dashboard"); - router.refresh(); + window.location.href = "/dashboard"; return; } setHasPassword(!!data.hasPassword); @@ -77,13 +76,12 @@ export default function LoginPage() { if (res.ok) { sessionStorage.setItem("omniroute_login_time", String(Date.now())); - router.push("/dashboard"); - router.refresh(); + window.location.href = "/dashboard"; } else { const data = await res.json(); // (#521) If no password is set, redirect to onboarding instead of showing an error if (data.needsSetup) { - router.push("/dashboard/onboarding"); + window.location.href = "/dashboard/onboarding"; return; } setError(data.error || t("invalidPassword")); diff --git a/tests/unit/login-11143.test.ts b/tests/unit/login-11143.test.ts new file mode 100644 index 00000000000..f55f2edecf3 --- /dev/null +++ b/tests/unit/login-11143.test.ts @@ -0,0 +1,21 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import fs from "node:fs"; +import path from "node:path"; + +test("login page performs full window.location navigation after authentication to avoid cookie race", () => { + const loginPagePath = path.resolve(process.cwd(), "src/app/login/page.tsx"); + const content = fs.readFileSync(loginPagePath, "utf8"); + + // Ensure router.push("/dashboard") is replaced with window.location.href + assert.equal( + content.includes('router.push("/dashboard")'), + false, + "LoginPage should not use router.push('/dashboard') after login" + ); + assert.equal( + content.includes('window.location.href = "/dashboard"'), + true, + "LoginPage must perform full window.location navigation after login" + ); +}); From 8fa3e314c8df9716b5a97569348c491bb2cec83f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Rouzbeh=E2=80=A0?= <78313022+rqzbeh@users.noreply.github.com> Date: Sun, 23 Aug 2026 07:35:08 +0330 Subject: [PATCH 22/29] fix(sse): unpin static Antigravity sessionId and add DNS retry classification (#10443) (#11177) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board + this branch: antigravity-dynamic-session-id + proxy-fetch-dns-retry green; file-size gate green with the proxyFetch 1244 frozen entry (dated annotation for the +5 retry-classification lines, owner-authorized). Static per-account sessionId unpinning ends the concurrent-turn 429s and EmptyStreamError drops on the Hermes→Antigravity path; EAI_AGAIN/ENOTFOUND/ETIMEDOUT now classified retryable. Conflict with the tip was only stale provider-count docs. Resolves the remaining #10443 root causes. Thank you @rqzbeh! --- config/quality/file-size-baseline.json | 3 ++- open-sse/services/antigravityIdentity.ts | 1 - open-sse/utils/proxyFetch.ts | 6 +++++ ...tigravity-dynamic-session-id-10443.test.ts | 18 +++++++++++++ .../unit/proxy-fetch-dns-retry-10443.test.ts | 27 +++++++++++++++++++ 5 files changed, 53 insertions(+), 2 deletions(-) create mode 100644 tests/unit/antigravity-dynamic-session-id-10443.test.ts create mode 100644 tests/unit/proxy-fetch-dns-retry-10443.test.ts diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 2eb468e16c6..28accfeae5b 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -441,7 +441,8 @@ "open-sse/executors/kiro.ts": 1390, "open-sse/translator/request/openai-to-kiro.ts": 1374, "open-sse/utils/sseHeartbeat.ts": 194, - "open-sse/utils/proxyFetch.ts": 1239, + "open-sse/utils/proxyFetch.ts": 1244, + "_rebaseline_2026_08_23_11177_dns_retry_classification": "PR #11177 (rqzbeh) own growth: proxyFetch.ts 1239->1244 (+5, EAI_AGAIN/ENOTFOUND/ETIMEDOUT join the retryable dispatcher classification alongside ECONNREFUSED — bounded socket retries for transient DNS failures, part of the #10443 Hermes→Antigravity stream-drop fixes). Covered by tests/unit/proxy-fetch-dns-retry-10443.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry: DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legítima acima do cap; gateways.ts = god-file de catálogo de providers que cresceu com os PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o próprio PR #9421 foi o que quebrou o arquivo; sem split até o release, congelado no tamanho atual). Owner autorizou rebaseline com anotação (2026-08-11).": { "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 1062, "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 1051, diff --git a/open-sse/services/antigravityIdentity.ts b/open-sse/services/antigravityIdentity.ts index f3934e84605..3703cd9c35b 100644 --- a/open-sse/services/antigravityIdentity.ts +++ b/open-sse/services/antigravityIdentity.ts @@ -75,7 +75,6 @@ export function getAntigravitySessionId( fallback?: unknown ): string { return ( - deriveAntigravitySessionId(getAntigravityAccountKey(credentials)) || toNonEmptyString(fallback) || generateAntigravitySessionId() ); diff --git a/open-sse/utils/proxyFetch.ts b/open-sse/utils/proxyFetch.ts index e8e3ea26f87..4eedd2dad56 100644 --- a/open-sse/utils/proxyFetch.ts +++ b/open-sse/utils/proxyFetch.ts @@ -858,6 +858,12 @@ async function patchedFetch( msg.includes("fetch failed") || errCode === "ECONNREFUSED" || msg.includes("ECONNREFUSED") || + errCode === "EAI_AGAIN" || + msg.includes("EAI_AGAIN") || + errCode === "ENOTFOUND" || + msg.includes("ENOTFOUND") || + errCode === "ETIMEDOUT" || + msg.includes("ETIMEDOUT") || (typeof errCode === "string" && errCode.startsWith("UND_ERR")) || msg.includes("UND_ERR") ) { diff --git a/tests/unit/antigravity-dynamic-session-id-10443.test.ts b/tests/unit/antigravity-dynamic-session-id-10443.test.ts new file mode 100644 index 00000000000..af5d198ff9b --- /dev/null +++ b/tests/unit/antigravity-dynamic-session-id-10443.test.ts @@ -0,0 +1,18 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { getAntigravitySessionId } from "../../open-sse/services/antigravityIdentity.ts"; + +test("getAntigravitySessionId yields dynamic random session IDs per request to avoid session pinning", () => { + const credentials = { email: "user@example.com", connectionId: "conn_123" }; + + const id1 = getAntigravitySessionId(credentials); + const id2 = getAntigravitySessionId(credentials); + + assert.notEqual(id1, id2, "getAntigravitySessionId should not pin to a static account email hash"); + assert.equal(typeof id1, "string"); + assert.equal(typeof id2, "string"); + + const explicitFallback = "custom-session-456"; + const idWithFallback = getAntigravitySessionId(credentials, explicitFallback); + assert.equal(idWithFallback, explicitFallback, "explicit fallback session ID should take precedence"); +}); diff --git a/tests/unit/proxy-fetch-dns-retry-10443.test.ts b/tests/unit/proxy-fetch-dns-retry-10443.test.ts new file mode 100644 index 00000000000..2518f58141a --- /dev/null +++ b/tests/unit/proxy-fetch-dns-retry-10443.test.ts @@ -0,0 +1,27 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; + +test("proxyFetch identifies transient DNS and network errors (EAI_AGAIN, ENOTFOUND, ECONNREFUSED) as retryable dispatcher errors", () => { + const isRetryableError = (err: unknown): boolean => { + const msg = err instanceof Error ? err.message : String(err); + const errCode = (err as { code?: unknown })?.code; + return Boolean( + msg.includes("fetch failed") || + errCode === "ECONNREFUSED" || + msg.includes("ECONNREFUSED") || + errCode === "EAI_AGAIN" || + msg.includes("EAI_AGAIN") || + errCode === "ENOTFOUND" || + msg.includes("ENOTFOUND") || + errCode === "ETIMEDOUT" || + msg.includes("ETIMEDOUT") || + (typeof errCode === "string" && errCode.startsWith("UND_ERR")) || + msg.includes("UND_ERR") + ); + }; + + assert.equal(isRetryableError({ code: "EAI_AGAIN", message: "getaddrinfo EAI_AGAIN www.googleapis.com" }), true); + assert.equal(isRetryableError({ code: "ENOTFOUND", message: "getaddrinfo ENOTFOUND api.example.com" }), true); + assert.equal(isRetryableError({ code: "ECONNREFUSED", message: "connect ECONNREFUSED 127.0.0.1:20128" }), true); + assert.equal(isRetryableError(new Error("HTTP 404 Not Found")), false); +}); From c92bd40b883a3d51736bd7086b13af4201d7fcd5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Rouzbeh=E2=80=A0?= <78313022+rqzbeh@users.noreply.github.com> Date: Sun, 23 Aug 2026 07:37:32 +0330 Subject: [PATCH 23/29] perf(proxy): implement non-blocking async proxy log batching and performance optimizations (A, B, C, D) (#11182) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board + this branch: perf-a-b-c-d + account-fallback-service 93/93, gates + typecheck clean. Owner approved the full bundle including item A (async proxy-log batching, 1s/100-item flush with an unref'd timer — reviewed the implementation: flush helper exists for shutdown wiring, buffered logs are the accepted tradeoff). B (lazy modals), C (O(1) alias maps), D (pre-compiled regex — this is also the entire content of #11187, being closed as subsumed) ride along. Conflict with the tip was only stale provider-count docs. Thank you @rqzbeh! --- open-sse/services/cloudCodeThinking.ts | 7 +- .../(dashboard)/dashboard/providers/page.tsx | 14 +- src/lib/proxyLogger.ts | 124 +++++++++++++----- tests/unit/perf-a-b-c-d.test.ts | 26 ++++ 4 files changed, 133 insertions(+), 38 deletions(-) create mode 100644 tests/unit/perf-a-b-c-d.test.ts diff --git a/open-sse/services/cloudCodeThinking.ts b/open-sse/services/cloudCodeThinking.ts index 443bc6510ec..b9c3e434a15 100644 --- a/open-sse/services/cloudCodeThinking.ts +++ b/open-sse/services/cloudCodeThinking.ts @@ -6,11 +6,10 @@ function isRecord(value: unknown): value is Record { return !!value && typeof value === "object" && !Array.isArray(value); } +const PREFIX_TRIM_RE = /^(?:models\/|antigravity\/)+/i; + function normalizeCloudCodeModel(model: string): string { - return String(model || "") - .trim() - .replace(/^models\//i, "") - .replace(/^antigravity\//i, ""); + return String(model || "").trim().replace(PREFIX_TRIM_RE, ""); } function stripGeminiThinkingConfig(value: unknown): unknown { diff --git a/src/app/(dashboard)/dashboard/providers/page.tsx b/src/app/(dashboard)/dashboard/providers/page.tsx index 6c853573b9c..548ca1514cf 100644 --- a/src/app/(dashboard)/dashboard/providers/page.tsx +++ b/src/app/(dashboard)/dashboard/providers/page.tsx @@ -46,9 +46,19 @@ import { getCodexGlobalServiceMode, type CodexGlobalServiceMode, } from "@/lib/providers/codexFastTier"; -import AddCompatibleProviderModal from "./components/AddCompatibleProviderModal"; +import dynamic from "next/dynamic"; +const AddCompatibleProviderModal = dynamic( + () => import("./components/AddCompatibleProviderModal"), + { ssr: false } +); import { CategoryDot } from "./components/CategoryDot"; -import { ImportProvidersFromFileModal } from "./components/ImportProvidersFromFileModal"; +const ImportProvidersFromFileModal = dynamic( + () => + import("./components/ImportProvidersFromFileModal").then( + (m) => m.ImportProvidersFromFileModal + ), + { ssr: false } +); import NoAuthProvidersSection from "./components/NoAuthProvidersSection"; import HighlightableProviderCard from "./components/HighlightableProviderCard"; import ProviderCountBadge from "./components/ProviderCountBadge"; diff --git a/src/lib/proxyLogger.ts b/src/lib/proxyLogger.ts index a9eb4b3805d..8665e20c758 100644 --- a/src/lib/proxyLogger.ts +++ b/src/lib/proxyLogger.ts @@ -194,43 +194,103 @@ export function logProxyEvent(entry: ProxyLogInput) { proxyLogs.length = MAX_IN_MEMORY_ENTRIES; } - // 2. Persist to SQLite + // 2. Queue for background batch persistence (SQLite / Redis) if (shouldPersistToDisk) { + enqueueProxyLog(log); + } + + return log; +} + +// ──────────────── Background Batch Persistence ──────────────── + +const BATCH_FLUSH_INTERVAL_MS = 1000; +const BATCH_SIZE_THRESHOLD = 100; + +let pendingLogsQueue: ProxyLogEntry[] = []; +let batchTimer: NodeJS.Timeout | null = null; + +function ensureBatchTimer() { + if (batchTimer) return; + batchTimer = setInterval(() => { + flushProxyLogsSync(); + }, BATCH_FLUSH_INTERVAL_MS); + if (typeof batchTimer.unref === "function") { + batchTimer.unref(); + } +} + +function enqueueProxyLog(log: ProxyLogEntry) { + pendingLogsQueue.push(log); + ensureBatchTimer(); + if (pendingLogsQueue.length >= BATCH_SIZE_THRESHOLD) { + flushProxyLogsSync(); + } +} + +export function flushProxyLogsSync() { + if (pendingLogsQueue.length === 0) return; + const batch = pendingLogsQueue; + pendingLogsQueue = []; + + // 1. If Redis driver is active, asynchronously publish batch to Redis Stream/Channel + if (process.env.QUOTA_STORE_DRIVER === "redis" || process.env.QUOTA_STORE_REDIS_URL) { try { - const db = getDbInstance(); - db.prepare( - `INSERT INTO proxy_logs (id, timestamp, status, proxy_type, proxy_host, proxy_port, - level, level_id, provider, target_url, public_ip, egress_ip, latency_ms, error, - connection_id, combo_id, account, tls_fingerprint) - VALUES (@id, @timestamp, @status, @proxyType, @proxyHost, @proxyPort, - @level, @levelId, @provider, @targetUrl, @clientIp, @egressIp, @latencyMs, @error, - @connectionId, @comboId, @account, @tlsFingerprint)` - ).run({ - id: log.id, - timestamp: log.timestamp, - status: log.status, - proxyType: log.proxy?.type || null, - proxyHost: log.proxy?.host || null, - proxyPort: log.proxy?.port ? Number(log.proxy.port) : null, - level: log.level, - levelId: log.levelId, - provider: log.provider, - targetUrl: log.targetUrl, - clientIp: log.clientIp, - egressIp: log.egressIp, - latencyMs: log.latencyMs, - error: log.error, - connectionId: log.connectionId, - comboId: log.comboId, - account: log.account, - tlsFingerprint: log.tlsFingerprint ? 1 : 0, - }); - } catch (err: any) { - console.warn("[proxyLogger] Failed to persist:", err.message); + import("@/lib/quota/redisQuotaStore").then(({ getRedisQuotaStore }) => { + const store = getRedisQuotaStore(process.env.QUOTA_STORE_REDIS_URL || ""); + const client = (store as any)?.client; + if (client && typeof client.publish === "function") { + for (const entry of batch) { + client.publish("omniroute:proxy_logs", JSON.stringify(entry)).catch(() => {}); + } + } + }).catch(() => {}); + } catch { + /* ignore redis pub errors */ } } - return log; + // 2. Persist to SQLite using a single transaction for high-performance non-blocking write + try { + const db = getDbInstance(); + const insertStmt = db.prepare( + `INSERT INTO proxy_logs (id, timestamp, status, proxy_type, proxy_host, proxy_port, + level, level_id, provider, target_url, public_ip, egress_ip, latency_ms, error, + connection_id, combo_id, account, tls_fingerprint) + VALUES (@id, @timestamp, @status, @proxyType, @proxyHost, @proxyPort, + @level, @levelId, @provider, @targetUrl, @clientIp, @egressIp, @latencyMs, @error, + @connectionId, @comboId, @account, @tlsFingerprint)` + ); + + const transaction = db.transaction((entries: ProxyLogEntry[]) => { + for (const item of entries) { + insertStmt.run({ + id: item.id, + timestamp: item.timestamp, + status: item.status, + proxyType: item.proxy?.type || null, + proxyHost: item.proxy?.host || null, + proxyPort: item.proxy?.port ? Number(item.proxy.port) : null, + level: item.level, + levelId: item.levelId, + provider: item.provider, + targetUrl: item.targetUrl, + clientIp: item.clientIp, + egressIp: item.egressIp, + latencyMs: item.latencyMs, + error: item.error, + connectionId: item.connectionId, + comboId: item.comboId, + account: item.account, + tlsFingerprint: item.tlsFingerprint ? 1 : 0, + }); + } + }); + + transaction(batch); + } catch (err: any) { + console.warn("[proxyLogger] Failed to write proxy log batch to disk:", err?.message || err); + } } // ──────────────── Query ──────────────── diff --git a/tests/unit/perf-a-b-c-d.test.ts b/tests/unit/perf-a-b-c-d.test.ts new file mode 100644 index 00000000000..0a131e0a34d --- /dev/null +++ b/tests/unit/perf-a-b-c-d.test.ts @@ -0,0 +1,26 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { logProxyEvent, flushProxyLogsSync } from "../../src/lib/proxyLogger.ts"; +import { shouldStripCloudCodeThinking } from "../../open-sse/services/cloudCodeThinking.ts"; + +test("Part A: async proxy log batching queues entries without synchronous failure", () => { + const sampleLog = { + status: "success", + provider: "test-provider", + latencyMs: 15, + }; + + const logged = logProxyEvent(sampleLog); + assert.equal(logged.provider, "test-provider"); + assert.equal(typeof logged.id, "string"); + + // Ensure flush completes without throwing + assert.doesNotThrow(() => { + flushProxyLogsSync(); + }); +}); + +test("Part D: pre-compiled regex in cloudCodeThinking model normalization", () => { + assert.equal(shouldStripCloudCodeThinking("antigravity", "antigravity/claude-3-7-sonnet"), true); + assert.equal(shouldStripCloudCodeThinking("antigravity", "models/gemini-2.5-pro"), false); +}); From 376b49d8a59ce6f11fb1820b7735353ef5df95bb Mon Sep 17 00:00:00 2001 From: Paco Cartones <253313177+pacocartones@users.noreply.github.com> Date: Sun, 23 Aug 2026 06:24:16 +0200 Subject: [PATCH 24/29] fix(ratelimit): keep operator minTime floor when relaxing on headroom (#9763) (#11086) Validated on the combined batch board over tip c92bd40b: static gates clean (changelog, file-size 158 frozen, complexity 2626<=2774, cognitive 1183<=1223, dead-code 409<=416), typecheck:core clean, 70 focused tests green (PR suites 49/49 + auth/combo neighbors 21/21). Operator-configured positive minTime floor now survives the plenty-of-headroom relaxation (resolveMinTime instead of a hard 0). Fixes #9763. Thank you @pacocartones! --- .../fixes/9763-ratelimit-mintime-floor.md | 1 + open-sse/services/rateLimitManager.ts | 2 +- ...ateLimitManager-mintime-floor-9763.test.ts | 88 +++++++++++++++++++ 3 files changed, 90 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/9763-ratelimit-mintime-floor.md create mode 100644 tests/unit/rateLimitManager-mintime-floor-9763.test.ts diff --git a/changelog.d/fixes/9763-ratelimit-mintime-floor.md b/changelog.d/fixes/9763-ratelimit-mintime-floor.md new file mode 100644 index 00000000000..2b145f1e1a3 --- /dev/null +++ b/changelog.d/fixes/9763-ratelimit-mintime-floor.md @@ -0,0 +1 @@ +- **fix(ratelimit):** respect operator `minTimeBetweenRequestsMs` floor when relaxing the limiter on headroom — the adaptive rate-limit learning no longer silently erases a configured minimum gap between requests when the upstream reports plenty of remaining capacity ([#9763](https://github.com/diegosouzapw/OmniRoute/issues/9763)). diff --git a/open-sse/services/rateLimitManager.ts b/open-sse/services/rateLimitManager.ts index a7bdeb2b09b..18815b32a5e 100644 --- a/open-sse/services/rateLimitManager.ts +++ b/open-sse/services/rateLimitManager.ts @@ -756,7 +756,7 @@ export function updateFromHeaders(provider, connectionId, headers, status, model ); } else if (remaining > limit * 0.5) { // Plenty of headroom — relax the limiter - updates.minTime = 0; + updates.minTime = resolveMinTime(currentRequestQueueSettings.minTimeBetweenRequestsMs); updates.reservoir = null; updates.reservoirRefreshAmount = null; updates.reservoirRefreshInterval = null; diff --git a/tests/unit/rateLimitManager-mintime-floor-9763.test.ts b/tests/unit/rateLimitManager-mintime-floor-9763.test.ts new file mode 100644 index 00000000000..2c4ee557552 --- /dev/null +++ b/tests/unit/rateLimitManager-mintime-floor-9763.test.ts @@ -0,0 +1,88 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const rlm = await import("../../open-sse/services/rateLimitManager.ts"); +const { + enableRateLimitProtection, + withRateLimit, + updateFromHeaders, + applyRequestQueueSettings, + __setLimiterFactoryForTests, + __resetRateLimitManagerForTests, +} = rlm; + +test.beforeEach(async () => { + await __resetRateLimitManagerForTests(); +}); + +test("headroom relaxation respects operator minTimeBetweenRequestsMs floor (#9763)", async () => { + // Apply an operator-configured minTime floor of 200ms + await applyRequestQueueSettings({ + minTimeBetweenRequestsMs: 200, + concurrentRequests: 0, + requestsPerMinute: 0, + maxWaitMs: 30000, + autoEnableApiKeyProvider: false, + }); + + let capturedMinTime: number | undefined; + + // Inject a fake limiter whose updateSettings captures the minTime. + // eslint-disable-next-line @typescript-eslint/no-explicit-any + const noop = (): any => undefined; + + __setLimiterFactoryForTests(() => { + const listeners: Record void>> = {}; + const fake = { + // eslint-disable-next-line @typescript-eslint/no-explicit-any + updateSettings(updates: Record) { + capturedMinTime = typeof updates.minTime === "number" ? updates.minTime : undefined; + return fake; + }, + on(event: string, fn: (...args: unknown[]) => void) { + (listeners[event] ??= []).push(fn); + return fake; + }, + schedule(arg0: unknown, arg1?: unknown) { + const fn = typeof arg1 === "function" ? arg1 : typeof arg0 === "function" ? arg0 : noop; + return fn(); + }, + disconnect() { + return Promise.resolve(); + }, + chain() { + return fake; + }, + counts() { + return { RECEIVED: 0, QUEUED: 0, RUNNING: 0, EXECUTING: 0 }; + }, + currentReservoir() { + return Promise.resolve(null); + }, + stop() { + return Promise.resolve(); + }, + }; + return fake; + }); + + enableRateLimitProtection("test-mintime-floor"); + + // Materialize the limiter with a dummy request + await withRateLimit("openai", "test-mintime-floor", "gpt-4", async () => "ok"); + + // Simulate a response with plenty of headroom: remaining=80 > limit*0.5=50 + const headers = new Headers({ + "x-ratelimit-limit-requests": "100", + "x-ratelimit-remaining-requests": "80", + }); + updateFromHeaders("openai", "test-mintime-floor", headers, 200, "gpt-4"); + + // The operator configured minTime=200, so headroom relaxation MUST NOT + // override it to 0. Before the fix, capturedMinTime === 0 (RED). + assert.strictEqual( + capturedMinTime, + 200, + `Expected minTime=200 (operator floor), got ${capturedMinTime}` + ); +}); From b6fc55911e431b594376e14ebe6bc4ee575ba39a Mon Sep 17 00:00:00 2001 From: Paco Cartones <253313177+pacocartones@users.noreply.github.com> Date: Sun, 23 Aug 2026 06:24:19 +0200 Subject: [PATCH 25/29] fix(providers): g4f.space sub-providers no longer advertise a free tier (#10071) (#11185) Validated on the combined batch board over tip c92bd40b: static gates clean (changelog, file-size 158 frozen, complexity 2626<=2774, cognitive 1183<=1223, dead-code 409<=416), typecheck:core clean, 70 focused tests green (PR suites 49/49 + auth/combo neighbors 21/21). Five g4f-* entries re-flagged hasFree:false with the live-probed 402 evidence (proof-of-work wall, member key still works); the two dissenting providers deliberately untouched, matching the chutes/aimlapi/yi precedent. Fixes #10071. Thank you @pacocartones and @chirag127 for the capture! --- ...-g4f-space-anonymous-tier-proof-of-work.md | 1 + .../constants/providers/apikey/gateways.ts | 33 +++++++++------- .../unit/discontinued-providers-2026.test.ts | 39 +++++++++++++++++++ tests/unit/g4f-space-gateway-6650.test.ts | 8 +++- 4 files changed, 64 insertions(+), 17 deletions(-) create mode 100644 changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md diff --git a/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md b/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md new file mode 100644 index 00000000000..47f8de84fd2 --- /dev/null +++ b/changelog.d/fixes/10071-g4f-space-anonymous-tier-proof-of-work.md @@ -0,0 +1 @@ +- **fix(providers):** the five g4f.space sub-providers (Groq, Gemini, Pollinations, Ollama, NVIDIA) no longer advertise a free tier — a keyless `POST /v1/chat/completions` now returns `402 insufficient_credits` behind a proof-of-work "cake" wall (re-verified live 2026-08-22), so `hasFree` is `false` and the notes point at `g4f.dev/members.html`. The gateway still works with a member key, so its registry wiring and `authType: "optional"` are unchanged ([#10071](https://github.com/diegosouzapw/OmniRoute/issues/10071)) — thanks @chirag127 diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts index 0b8d07759f3..6dab614425e 100644 --- a/src/shared/constants/providers/apikey/gateways.ts +++ b/src/shared/constants/providers/apikey/gateways.ts @@ -673,11 +673,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, - freeNote: "Free no-key reverse proxy to Groq (gpt4free project) — rate-limited to 5 req/min.", + hasFree: false, + freeNote: + "No-key reverse proxy to Groq (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-gemini": { id: "g4f-gemini", @@ -687,11 +688,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, - freeNote: "Free no-key reverse proxy to Gemini (gpt4free project) — rate-limited to 5 req/min.", + hasFree: false, + freeNote: + "No-key reverse proxy to Gemini (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-pollinations": { id: "g4f-pollinations", @@ -701,12 +703,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, + hasFree: false, freeNote: - "Free no-key reverse proxy to Pollinations (gpt4free project) — rate-limited to 5 req/min.", + "No-key reverse proxy to Pollinations (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-ollama": { id: "g4f-ollama", @@ -716,11 +718,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, - freeNote: "Free no-key hosted Ollama gateway (gpt4free project) — rate-limited to 5 req/min.", + hasFree: false, + freeNote: + "No-key hosted Ollama gateway (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "g4f-nvidia": { id: "g4f-nvidia", @@ -730,12 +733,12 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#F97316", textIcon: "G4F", website: "https://g4f.space", - hasFree: true, + hasFree: false, freeNote: - "Free no-key reverse proxy to NVIDIA NIM (gpt4free project) — rate-limited to 5 req/min.", + "No-key reverse proxy to NVIDIA NIM (gpt4free project) — the anonymous free tier is gone; keyless calls return insufficient_credits until you bake proof-of-work credits. A g4f.dev member key is required.", passthroughModels: true, authHint: - "No auth required. Free tier is limited to 5 requests/minute — sign up at g4f.dev/members.html for higher limits.", + "Anonymous use now needs proof-of-work credits baked at g4f.dev/chat — sign up at g4f.dev/members.html for a member key.", }, "vercel-ai-gateway": { id: "vercel-ai-gateway", diff --git a/tests/unit/discontinued-providers-2026.test.ts b/tests/unit/discontinued-providers-2026.test.ts index 1979a7cd9aa..88f963ef911 100644 --- a/tests/unit/discontinued-providers-2026.test.ts +++ b/tests/unit/discontinued-providers-2026.test.ts @@ -6,6 +6,9 @@ import assert from "node:assert"; // free tier that does not exist. The budget catalog already dropped them. The 2026-06-18 batch // (gitlawb, gitlawb-gmi, aimlapi, yi) was each re-verified against the official source before flipping // (aimlapi docs: "The Free Tier is currently paused"; gitlawb GitHub issue #1345: MiMo revoked). +// 2026-08-22 (#10071): the five g4f.space sub-providers lost their anonymous tier to a proof-of-work +// credit wall (keyless POST -> HTTP 402 insufficient_credits). They remain usable with a g4f.dev +// member key, so only hasFree/freeNote/authHint changed - registry wiring is untouched. describe("2026 discontinued free tiers — providers.ts hasFree reconciliation", () => { it("APIKEY_PROVIDERS dead tiers no longer advertise a free tier", async () => { const { APIKEY_PROVIDERS } = await import("../../src/shared/constants/providers.ts"); @@ -27,6 +30,42 @@ describe("2026 discontinued free tiers — providers.ts hasFree reconciliation", } }); + it("g4f.space sub-providers no longer advertise an anonymous free tier", async () => { + const { APIKEY_PROVIDERS } = await import("../../src/shared/constants/providers.ts"); + // 2026-08-22 live re-verification: every g4f.space sub-path still lists models keylessly, but a + // keyless POST /v1/chat/completions returns HTTP 402 {"type":"insufficient_credits"} pointing at + // a proof-of-work "cake" wall (g4f.dev/chat) or a member key (g4f.dev/members.html). The gateway + // is NOT dead - it works with a g4f.dev member key - so the registry entries and their + // authType:"optional" are deliberately untouched; only the free-tier advertisement is corrected. + for (const id of ["g4f-groq", "g4f-gemini", "g4f-pollinations", "g4f-ollama", "g4f-nvidia"]) { + const p = ( + APIKEY_PROVIDERS as Record< + string, + { hasFree?: boolean; freeNote?: string; authHint?: string } + > + )[id]; + assert.ok( + p, + `${id} should still exist in APIKEY_PROVIDERS (gateway still usable with a member key)` + ); + assert.strictEqual( + p.hasFree, + false, + `${id} should have hasFree:false (anonymous tier walled behind proof-of-work credits in 2026)` + ); + assert.match( + p.freeNote ?? "", + /proof-of-work/i, + `${id} freeNote should explain the proof-of-work credit wall` + ); + assert.match( + p.authHint ?? "", + /member key/i, + `${id} authHint should state that a g4f.dev member key is required` + ); + } + }); + it("phind is fully removed (service shut down 2026-01) from both catalogs", async () => { const { APIKEY_PROVIDERS, WEB_COOKIE_PROVIDERS } = await import("../../src/shared/constants/providers.ts"); diff --git a/tests/unit/g4f-space-gateway-6650.test.ts b/tests/unit/g4f-space-gateway-6650.test.ts index 8f1aadc2bcd..93fd99bd679 100644 --- a/tests/unit/g4f-space-gateway-6650.test.ts +++ b/tests/unit/g4f-space-gateway-6650.test.ts @@ -17,7 +17,8 @@ * category on the dashboard * - allowed to skip API key validation (providerAllowsOptionalApiKey) * - has provider metadata (name/website/free-tier note) in the apikey - * gateway catalog + * gateway catalog (hasFree flipped false by #10071 — anonymous tier now + * requires proof-of-work credits; a g4f.dev member key is required) */ import test from "node:test"; import assert from "node:assert/strict"; @@ -96,7 +97,10 @@ for (const [id, subPath] of Object.entries(SUB_PATHS)) { assert.ok(meta, `${id} should have an APIKEY_PROVIDERS metadata entry`); assert.equal(meta.id, id); assert.equal(meta.website, "https://g4f.space"); - assert.equal(meta.hasFree, true); + // hasFree was true at #6650 time; the anonymous tier was walled behind proof-of-work + // credits in 2026 (#10071), so the flag is now false. Registry wiring above is unchanged: + // the provider still works with a g4f.dev member key, hence authType stays "optional". + assert.equal(meta.hasFree, false); assert.equal(typeof meta.freeNote, "string"); assert.ok((meta.freeNote as string).length > 0); }); From 8c7338651c4593c390296cc6e26e11e05b5f0f2d Mon Sep 17 00:00:00 2001 From: Paco Cartones <253313177+pacocartones@users.noreply.github.com> Date: Sun, 23 Aug 2026 06:24:22 +0200 Subject: [PATCH 26/29] fix(routing): forward lkgpEnabled into RoutingContext so the LKGP toggle works (#11193) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board over tip c92bd40b: static gates clean (changelog, file-size 158 frozen, complexity 2626<=2774, cognitive 1183<=1223, dead-code 409<=416), typecheck:core clean, 70 focused tests green (PR suites 49/49 + auth/combo neighbors 21/21). lkgpEnabled finally reaches the RoutingContext literal — the Settings→Routing LKGP toggle was persisted but unreachable (context.lkgpEnabled always undefined). Scope discipline noted and appreciated: the applyStrategyOrdering dependency change stays out. Fixes #11181. Thank you @pacocartones! --- .../fixes/11181-lkgp-enabled-context.md | 1 + .../services/combo/resolveAutoStrategy.ts | 6 + tests/unit/lkgp-enabled-context-11181.test.ts | 141 ++++++++++++++++++ 3 files changed, 148 insertions(+) create mode 100644 changelog.d/fixes/11181-lkgp-enabled-context.md create mode 100644 tests/unit/lkgp-enabled-context-11181.test.ts diff --git a/changelog.d/fixes/11181-lkgp-enabled-context.md b/changelog.d/fixes/11181-lkgp-enabled-context.md new file mode 100644 index 00000000000..d1c0cde5a36 --- /dev/null +++ b/changelog.d/fixes/11181-lkgp-enabled-context.md @@ -0,0 +1 @@ +- **fix(routing):** the Routing tab's "last known good provider" toggle now actually takes effect — `lkgpEnabled` was persisted and the `lkgp` strategy guarded on it, but the setting was never forwarded into the `RoutingContext` built in `resolveAutoStrategyOrder()`, so `context.lkgpEnabled` was always `undefined` and the off-switch was unreachable ([#11181](https://github.com/diegosouzapw/OmniRoute/issues/11181)) diff --git a/open-sse/services/combo/resolveAutoStrategy.ts b/open-sse/services/combo/resolveAutoStrategy.ts index 7d6fe2a682d..45ef4b5c605 100644 --- a/open-sse/services/combo/resolveAutoStrategy.ts +++ b/open-sse/services/combo/resolveAutoStrategy.ts @@ -311,6 +311,12 @@ export async function resolveAutoStrategyOrder( taskType, requestHasTools, lastKnownGoodProvider, + // #11181: the Routing tab persists an LKGP on/off toggle and + // LKGPStrategy guards on `context.lkgpEnabled === false`, but the + // field was never forwarded into this context, so the guard never + // saw the setting and the off-switch was unreachable. + lkgpEnabled: (settings as { lkgpEnabled?: unknown } | null | undefined)?.lkgpEnabled as + boolean | undefined, estimatedInputTokens, sla: slaPolicy, }, diff --git a/tests/unit/lkgp-enabled-context-11181.test.ts b/tests/unit/lkgp-enabled-context-11181.test.ts new file mode 100644 index 00000000000..04950c2c2a5 --- /dev/null +++ b/tests/unit/lkgp-enabled-context-11181.test.ts @@ -0,0 +1,141 @@ +/** + * #11181 — the `lkgpEnabled` settings toggle must actually reach RoutingContext. + * + * `LKGPStrategyImpl.select()` guards with `context.lkgpEnabled === false` + * (open-sse/services/autoCombo/routerStrategy.ts), and the setting is persisted + * by the Routing settings tab (src/shared/validation/settingsSchemas.ts). But the + * RoutingContext literal built in resolveAutoStrategyOrder() never carried the + * field, so the guard could never fire in production. + * + * tests/unit/router-strategies.test.ts already covers the guard — but it hands + * the strategy a context it built itself, so it stays green whether or not the + * production construction site populates the field. These tests drive + * resolveAutoStrategyOrder() with a *persisted* setting instead, which is the + * only level at which the wiring is observable. + */ +import { test, after } from "node:test"; +import assert from "node:assert/strict"; + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-lkgp-11181-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const { resolveAutoStrategyOrder } = + await import("@omniroute/open-sse/services/combo/resolveAutoStrategy.ts"); +const settingsDb = await import("@/lib/db/settings.ts"); +const { resetDbInstance } = await import("@/lib/db/core.ts"); + +after(() => { + resetDbInstance(); +}); + +const target = (provider: string, modelStr: string): never => + ({ + kind: "model", + stepId: "s1", + executionKey: `${provider}>${modelStr}`, + modelStr, + provider, + providerId: null, + connectionId: null, + weight: 1, + label: null, + }) as never; + +const candidate = (provider: string, model: string, overrides: Record = {}) => ({ + kind: "model", + stepId: "s1", + executionKey: `${provider}>${model}`, + modelStr: model, + provider, + model, + quotaRemaining: 100, + quotaTotal: 100, + circuitBreakerState: "CLOSED", + costPer1MTokens: 1, + p95LatencyMs: 100, + latencyStdDev: 10, + errorRate: 0, + ...overrides, +}); + +// "cheap" wins under the rules scorer (cheapest + fastest + most stable); +// "pricey" only ever wins by being the persisted last-known-good provider. +const candidates = () => + [ + candidate("openai", "cheap-model", { + costPer1MTokens: 0.01, + p95LatencyMs: 10, + latencyStdDev: 1, + }), + candidate("anthropic", "pricey-model", { + costPer1MTokens: 50, + p95LatencyMs: 5000, + latencyStdDev: 900, + }), + ] as never; + +function capturingLog() { + const entries: string[] = []; + const push = (_tag: unknown, msg: unknown) => entries.push(String(msg)); + return { entries, info: push, warn: push, error: push, debug: push }; +} + +async function runWithSettings(comboName: string, settings: Record | null) { + // The LKGP pin resolveAutoStrategyOrder reads is getLKGP(combo.name, combo.id || combo.name). + await settingsDb.setLKGP(comboName, comboName, "anthropic"); + + const log = capturingLog(); + const result = await resolveAutoStrategyOrder({ + orderedTargets: [target("openai", "cheap-model"), target("anthropic", "pricey-model")], + body: { messages: [{ role: "user", content: "hi" }] }, + combo: { + id: comboName, + name: comboName, + autoConfig: { + routerStrategy: "lkgp", + candidatePool: ["openai", "anthropic"], + explorationRate: 0, + }, + }, + settings, + config: {}, + relayOptions: null, + resilienceSettings: { quotaPreflight: { enabled: false } }, + log, + buildAutoCandidates: (async () => candidates()) as never, + } as never); + + assert.ok("orderedTargets" in result, "expected an ordering result, not an earlyResponse"); + const selection = log.entries.find((entry) => entry.startsWith("Auto selection:")) ?? ""; + return { result, selection }; +} + +test("control — with lkgpEnabled unset the LKGP pin still wins (guard must not over-fire)", async () => { + const { result, selection } = await runWithSettings("lkgp-11181-default", null); + assert.match(selection, /LKGP: using last known good provider anthropic/); + if ("orderedTargets" in result) { + assert.equal(result.orderedTargets[0].provider, "anthropic"); + } +}); + +test("#11181 — a persisted lkgpEnabled:false makes the lkgp strategy delegate to rules", async () => { + const { result, selection } = await runWithSettings("lkgp-11181-disabled", { + lkgpEnabled: false, + }); + + // The whole point of the toggle: the LKGP pin must be ignored and the 6-factor + // rules scorer must pick the winner instead. + assert.doesNotMatch( + selection, + /LKGP: using last known good provider/, + `lkgpEnabled:false must disable LKGP selection, got: ${selection}` + ); + assert.match(selection, /RulesStrategy: score=/, `expected rules fallback, got: ${selection}`); + if ("orderedTargets" in result) { + assert.equal(result.orderedTargets[0].provider, "openai"); + } +}); From a966c7520b9b139597ab132f2c4a6a0b3bd975d3 Mon Sep 17 00:00:00 2001 From: Paco Cartones <253313177+pacocartones@users.noreply.github.com> Date: Sun, 23 Aug 2026 06:24:25 +0200 Subject: [PATCH 27/29] fix(routing): keep keyless custom-compatible connections in the auto/* pool (#11198) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board over tip c92bd40b: static gates clean (changelog, file-size 158 frozen, complexity 2626<=2774, cognitive 1183<=1223, dead-code 409<=416), typecheck:core clean, 70 focused tests green (PR suites 49/49 + auth/combo neighbors 21/21). Keyless custom-compatible connections stay in auto/* pools — the credential filter now recognizes a registry-free keyless endpoint instead of dropping the connection before pool construction. Fixes #11180. Thank you @pacocartones! --- ...11180-keyless-custom-provider-auto-pool.md | 1 + open-sse/services/autoCombo/virtualFactory.ts | 25 ++++- ...auto-keyless-custom-provider-11180.test.ts | 91 +++++++++++++++++++ 3 files changed, 116 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md create mode 100644 tests/unit/auto-keyless-custom-provider-11180.test.ts diff --git a/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md b/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md new file mode 100644 index 00000000000..c533feae543 --- /dev/null +++ b/changelog.d/fixes/11180-keyless-custom-provider-auto-pool.md @@ -0,0 +1 @@ +- **fix(routing):** a custom `openai-compatible-*` / `anthropic-compatible-*` connection pointing at a keyless self-hosted backend (llama.cpp, Ollama, vLLM started without an API key) now stays in the `auto/*` candidate pool instead of being silently dropped by the credential gate — for those IDs "no credential" is the normal configuration, not an unconfigured connection ([#11180](https://github.com/diegosouzapw/OmniRoute/pull/11180)) — thanks @marcs7 diff --git a/open-sse/services/autoCombo/virtualFactory.ts b/open-sse/services/autoCombo/virtualFactory.ts index 93d4a0b53e9..e3dbac4d73c 100644 --- a/open-sse/services/autoCombo/virtualFactory.ts +++ b/open-sse/services/autoCombo/virtualFactory.ts @@ -8,6 +8,7 @@ import type { ConnectionFields } from "@/lib/db/encryption"; import { NOAUTH_PROVIDERS } from "@/shared/constants/providers"; import { hasUsableWebSessionCredential } from "@/shared/providers/webSessionCredentials"; import { toNumber } from "@/shared/utils/numeric"; +import { isCompatibleProviderConnectionId } from "@/shared/utils/compatibleProviderId"; import { defaultLogger as log } from "@omniroute/open-sse/utils/logger"; import { getTokenLimit } from "../contextManager"; import { @@ -180,9 +181,31 @@ function hasProviderSpecificSessionData(conn: VirtualFactoryConn): boolean { return hasUsableWebSessionCredential(conn.provider, conn.providerSpecificData); } +/** + * #11180: a custom compatible connection (`openai-compatible-*` / + * `anthropic-compatible-*`) may legitimately carry no credential at all, + * because it points at a self-hosted backend the operator started without one + * (`llama-server --host 0.0.0.0` with no `--api-key`, Ollama, vLLM). For those + * IDs "no credential" is the normal configuration rather than an unconfigured + * connection, so the credential gate must not silently drop them from every + * `auto/*` pool while direct `/` calls keep working. + * + * Deliberately narrow: only the four generated compatible-provider ID shapes + * qualify. A first-party provider with an empty key really is unconfigured and + * stays filtered out, and the no-auth registry allowlist below is untouched. + */ +function isKeylessEligibleConnection(conn: VirtualFactoryConn): boolean { + return isCompatibleProviderConnectionId(conn.provider); +} + function hasUsableConnectionCredential(conn: VirtualFactoryConn): boolean { const hasApiKey = typeof conn.apiKey === "string" && conn.apiKey.trim().length > 0; - return hasApiKey || hasUsableOAuthToken(conn) || hasProviderSpecificSessionData(conn); + return ( + hasApiKey || + hasUsableOAuthToken(conn) || + hasProviderSpecificSessionData(conn) || + isKeylessEligibleConnection(conn) + ); } const SYNTHETIC_NOAUTH_CONNECTION_ID = RESILIENCE_NOAUTH_CONNECTION_ID; diff --git a/tests/unit/auto-keyless-custom-provider-11180.test.ts b/tests/unit/auto-keyless-custom-provider-11180.test.ts new file mode 100644 index 00000000000..c2e97e8158d --- /dev/null +++ b/tests/unit/auto-keyless-custom-provider-11180.test.ts @@ -0,0 +1,91 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// #11180 regression guard: a custom OpenAI-compatible connection pointing at a +// keyless local backend (llama.cpp / Ollama / vLLM started without an API key) +// carries no apiKey, no OAuth token and no provider-specific session data, so +// `hasUsableConnectionCredential` dropped it from `validConnections` before the +// auto/* candidate pool was built. The connection was active, tested and synced, +// yet structurally invisible to auto-routing with no log line and no UI hint. +// +// Keyless is the NORMAL configuration for a self-hosted backend, so a custom +// compatible connection must stay eligible. This gate is one step later than +// #5873 (registry-absent defaultModel fallback), whose guard still passes. + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-auto-keyless-11180-")); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; + +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const virtualFactory = await import("../../open-sse/services/autoCombo/virtualFactory.ts"); + +type VirtualComboResult = Awaited>; + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + await resetStorage(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + + if (ORIGINAL_DATA_DIR === undefined) { + delete process.env.DATA_DIR; + } else { + process.env.DATA_DIR = ORIGINAL_DATA_DIR; + } +}); + +test("keyless custom openai-compatible connection enters the auto pool (#11180)", async () => { + const customProvider = "openai-compatible-chat-c2fe8a44-f2fd-47b4-8893-6f1521804c45"; + await providersDb.createProviderConnection({ + provider: customProvider, + authType: "apikey", + name: "llamaAsimov", + // Keyless local backend: llama-server --host 0.0.0.0 with no --api-key. + apiKey: "", + defaultModel: "Qwen3.8-27B-UD-Q4-DFlash-GGUF", + }); + + const combo: VirtualComboResult = await virtualFactory.createVirtualAutoCombo("fast"); + + const candidate = combo.models.find((model) => model.providerId === customProvider); + assert.ok( + candidate, + "a keyless custom-compatible connection must not be dropped by the credential gate" + ); + assert.equal(candidate.model, `${customProvider}/Qwen3.8-27B-UD-Q4-DFlash-GGUF`); + assert.ok(combo.autoConfig.candidatePool.includes(customProvider)); +}); + +test("a keyless FIRST-PARTY provider connection stays out of the pool (#11180)", async () => { + // The relaxation is scoped to custom compatible connection IDs. A first-party + // provider with an empty key is an unconfigured connection, not a keyless + // local backend, and must still be filtered out. + await providersDb.createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "unconfigured openai", + apiKey: "", + defaultModel: "gpt-4o", + }); + + const combo: VirtualComboResult = await virtualFactory.createVirtualAutoCombo("fast"); + + assert.equal( + combo.autoConfig.candidatePool.includes("openai"), + false, + "an unconfigured first-party connection must remain excluded" + ); +}); From 8d6870f96e5f58c217e52dbce867d920dd4ae1d9 Mon Sep 17 00:00:00 2001 From: Paco Cartones <253313177+pacocartones@users.noreply.github.com> Date: Sun, 23 Aug 2026 06:24:28 +0200 Subject: [PATCH 28/29] fix(analytics): classify opencode-go as a flat-rate subscription (#11199) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated on the combined batch board over tip c92bd40b: static gates clean (changelog, file-size 158 frozen, complexity 2626<=2774, cognitive 1183<=1223, dead-code 409<=416), typecheck:core clean, 70 focused tests green (PR suites 49/49 + auth/combo neighbors 21/21). opencode-go joins FLAT_RATE_SUBSCRIPTION_PROVIDER_IDS — cost analytics stop pricing a flat 0 subscription at metered aggregator rates (3.35 reported vs 0 actual). Same pattern as #10774. Fixes #11149. Thank you @pacocartones! --- .../fixes/11149-opencode-go-flat-rate.md | 1 + src/lib/usage/flatRateProviders.ts | 5 +++++ tests/unit/flat-rate-cost-5552.test.ts | 22 +++++++++++++++++++ 3 files changed, 28 insertions(+) create mode 100644 changelog.d/fixes/11149-opencode-go-flat-rate.md diff --git a/changelog.d/fixes/11149-opencode-go-flat-rate.md b/changelog.d/fixes/11149-opencode-go-flat-rate.md new file mode 100644 index 00000000000..7aa63ad4559 --- /dev/null +++ b/changelog.d/fixes/11149-opencode-go-flat-rate.md @@ -0,0 +1 @@ +- **fix(analytics):** `opencode-go` is now classified as a flat-rate subscription, so cost analytics shows $0 for it instead of billing every call at the underlying model’s metered rate — it resells GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x under one flat monthly fee, which made the overstatement large rather than marginal ([#11149](https://github.com/diegosouzapw/OmniRoute/pull/11149)) — thanks @electrumguy diff --git a/src/lib/usage/flatRateProviders.ts b/src/lib/usage/flatRateProviders.ts index 8d3eec2c5d7..3454b9b3917 100644 --- a/src/lib/usage/flatRateProviders.ts +++ b/src/lib/usage/flatRateProviders.ts @@ -46,6 +46,11 @@ const FLAT_RATE_SUBSCRIPTION_PROVIDER_IDS: ReadonlySet = new Set([ "glm-cn", // GLM Coding (China) plan "claude", // Claude Code plan (OAuth-only — a Claude Pro/Max subscription) "cc", // Claude Code plan (alias id — same connection, shares the `cc` pricing rows) + // OpenCode Go subscription (https://opencode.ai/go) — a flat monthly fee. It is an + // aggregator reselling GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x, so + // per-token rows price each call at the UNDERLYING model's metered rate and the + // analytics overstatement is large rather than marginal (#11149). + "opencode-go", ]); /** diff --git a/tests/unit/flat-rate-cost-5552.test.ts b/tests/unit/flat-rate-cost-5552.test.ts index 2df378ee3e5..61ad9847b16 100644 --- a/tests/unit/flat-rate-cost-5552.test.ts +++ b/tests/unit/flat-rate-cost-5552.test.ts @@ -25,6 +25,7 @@ test("isFlatRateProvider: dedicated subscription / coding-plan providers are fla "glm-cn", "claude", "cc", + "opencode-go", ]) { assert.equal(isFlatRateProvider(id), true, `${id} should be flat-rate`); } @@ -83,6 +84,27 @@ test("computeCostFromPricing: opt-in only — flat-rate provider WITHOUT the fla assert.equal(computeCostFromPricing(PRICING, TOKENS, { provider: "chatgpt-web" }), 3); }); +test("#11149: opencode-go is a flat-rate subscription, not metered", () => { + // opencode-go (https://opencode.ai/go) is a $10/month flat subscription that + // resells GLM, Kimi, Grok, DeepSeek, MiniMax, Qwen and GPT-5.x. Because it is + // an aggregator, every call was priced at the UNDERLYING model's metered rate, + // so the overstatement is large rather than marginal (a reported ~$13.35 for a + // month actually billed at $10 flat). It is api-key auth, so it is not covered + // by the dynamic WEB_COOKIE_PROVIDERS branch and needs the explicit id. + assert.equal(isFlatRateProvider("opencode-go"), true); + assert.equal( + computeCostFromPricing(PRICING, TOKENS, { provider: "opencode-go", flatRateAsZero: true }), + 0 + ); + // Still opt-in: without the flag the per-request estimate is unchanged. + assert.equal(computeCostFromPricing(PRICING, TOKENS, { provider: "opencode-go" }), 3); +}); + +test("#11149: sibling opencode ids keep their own billing semantics", () => { + // Only the Go subscription is flat-rate. The keyless `opencode` provider is a + // different id and must not be swept in by a prefix-style match. + assert.equal(isFlatRateProvider("opencode"), false); +}); test("computeCostFromPricing: metered provider with the flag still estimates", () => { assert.equal( computeCostFromPricing(PRICING, TOKENS, { provider: "openai", flatRateAsZero: true }), From 1722c83726fb1909f5d0b7b0513896906ecbc09f Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Sun, 23 Aug 2026 01:25:18 -0300 Subject: [PATCH 29/29] chore(quality): freeze auth.ts at 3337 with annotation (#11186) --- config/quality/file-size-baseline.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 28accfeae5b..0a3239ea7fb 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -432,7 +432,8 @@ "src/shared/components/analytics/charts.tsx": 1346, "src/shared/services/cliRuntime.ts": 1459, "src/sse/handlers/chat.ts": 2493, - "src/sse/services/auth.ts": 3260, + "src/sse/services/auth.ts": 3337, + "_rebaseline_2026_08_23_11186_synced_inventory_routing": "PR #11186 (pacocartones) own growth: src/sse/services/auth.ts 3260->3337 (+77, loadAdvertisedModelsForSelfHostedConnections + the modelNotAdvertised candidate-filter predicate — pins chat routing to the connection whose synced inventory actually advertises the model, fixing spurious model-not-found on multi-host self-hosted setups; at the existing credential-selection chokepoint, not extractable without splitting the selection flow). Covered by tests/unit/chat-routing-synced-inventory-11089.test.ts. Owner pre-authorized baseline bumps 2026-08-22.", "tests/unit/account-fallback-service.test.ts": 2044, "tests/unit/provider-validation-specialty.test.ts": 3880, "open-sse/executors/hyperagent.ts": 1334,