diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 36f82549779..d12585363a6 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -22,10 +22,10 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: persist-credentials: false - - uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + - uses: github/codeql-action/init@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7 with: languages: javascript-typescript queries: security-extended - - uses: github/codeql-action/analyze@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + - uses: github/codeql-action/analyze@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7 with: category: "/language:javascript-typescript" diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index 6e4adc01929..d8a65576dc6 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -372,7 +372,7 @@ jobs: - name: Upload Trivy SARIF to Security tab if: needs.prepare.outputs.version != 'main' continue-on-error: true - uses: github/codeql-action/upload-sarif@v4.37.6 + uses: github/codeql-action/upload-sarif@v4.37.7 with: sarif_file: trivy-results.sarif category: trivy-image diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml index 33708fc426a..e899a664eaa 100644 --- a/.github/workflows/electron-release.yml +++ b/.github/workflows/electron-release.yml @@ -279,9 +279,14 @@ jobs: - name: Smoke packaged Electron app (Linux) if: matrix.platform == 'linux' + # #7592: also cold-restart against the same DATA_DIR and assert a + # native SQLite driver (not the sql.js WASM fallback) is selected on + # the second launch — blocking here since Linux has no Windows-style + # sandbox caveats that would make it flaky. env: ELECTRON_SMOKE_TIMEOUT_MS: 60000 ELECTRON_SMOKE_STREAM_LOGS: "1" + ELECTRON_SMOKE_COLD_RESTART: "1" run: xvfb-run -a npm run electron:smoke:packaged - name: Collect installers diff --git a/.gitignore b/.gitignore index a21784f4aae..08bceafd368 100644 --- a/.gitignore +++ b/.gitignore @@ -291,3 +291,4 @@ docker-compose.yml.bak # Ad-hoc test sandboxes (never tracked — may contain local DBs) /.sandbox/ +.aider* diff --git a/@omniroute/opencode-plugin/src/index.ts b/@omniroute/opencode-plugin/src/index.ts index 7c196aeccbe..be985361c95 100644 --- a/@omniroute/opencode-plugin/src/index.ts +++ b/@omniroute/opencode-plugin/src/index.ts @@ -76,6 +76,17 @@ import { type FreeModelFreeType, } from "./naming.js"; +/** + * Minimal leveled logger sink accepted by the default fetchers and the static + * catalog builder. A full `Logger` satisfies it structurally; the config hook + * injects the same partial shape (see `createOmniRouteConfigHook` deps). + */ +type OmniRouteLoggerSink = { + error?: (message: string, ...args: unknown[]) => void; + warn: (message: string, ...args: unknown[]) => void; + debug?: (message: string, ...args: unknown[]) => void; +}; + /** * Zod schema for plugin options accepted as the second element of the * `plugin: [name, opts]` tuple in opencode.json. Strict by design — unknown @@ -791,13 +802,18 @@ export async function forceSyncOmniRouteModels(args: { try { rawCombos = await combosFetcher(auth.baseURL, auth.managementReadToken, 10_000); } catch (err) { - console.warn("[omniroute-plugin] force sync: combos fetch failed", err); + logger.warn("force sync: combos fetch failed", err); } } let rawAutoCombos: OmniRouteRawAutoCombo[] = []; if (wantAutoCombos) { try { - rawAutoCombos = await autoCombosFetcher(auth.baseURL, auth.managementReadToken, 5_000); + rawAutoCombos = await autoCombosFetcher( + auth.baseURL, + auth.managementReadToken, + 5_000, + logger + ); } catch { /* soft-fail */ } @@ -1089,7 +1105,7 @@ export const OmniRoutePlugin: Plugin = async (_input, options) => { return { auth: createOmniRouteAuthHook(resolved), - provider: createOmniRouteProviderHook(resolved, { cache: sharedCache }), + provider: createOmniRouteProviderHook(resolved, { cache: sharedCache, logger }), config: configWithSyncCommand, tool: { omniroute_sync_models: syncTool, @@ -1676,7 +1692,8 @@ export interface OmniRouteRawAutoCombo { export type OmniRouteAutoCombosFetcher = ( baseURL: string, apiKey: string, - timeoutMs?: number + timeoutMs?: number, + logger?: OmniRouteLoggerSink ) => Promise; /** @@ -1688,9 +1705,11 @@ export type OmniRouteAutoCombosFetcher = ( export const defaultOmniRouteAutoCombosFetcher: OmniRouteAutoCombosFetcher = async ( baseURL, apiKey, - timeoutMs = 5_000 + timeoutMs = 5_000, + logger?: OmniRouteLoggerSink ) => { if (!apiKey || !baseURL) return []; + const log = logger ?? _logger; const trimmed = trimTrailingSlashes(baseURL); const root = trimmed.replace(/\/v\d+$/, ""); @@ -1709,15 +1728,11 @@ export const defaultOmniRouteAutoCombosFetcher: OmniRouteAutoCombosFetcher = asy }); // 404 = endpoint not deployed yet — expected during rollout if (res.status === 404) { - console.warn( - `[omniroute-plugin] /api/combos/auto not available (404) — auto combos disabled` - ); + log.warn(`/api/combos/auto not available (404) — auto combos disabled`); return []; } if (!res.ok) { - console.warn( - `[omniroute-plugin] /api/combos/auto failed: ${res.status} ${res.statusText} — auto combos disabled` - ); + log.warn(`/api/combos/auto failed: ${res.status} ${res.statusText} — auto combos disabled`); return []; } const body = (await res.json()) as unknown; @@ -1735,8 +1750,8 @@ export const defaultOmniRouteAutoCombosFetcher: OmniRouteAutoCombosFetcher = asy return out; } catch (err) { // Network error, timeout, abort — all non-fatal - console.warn( - `[omniroute-plugin] /api/combos/auto fetch failed: ${err instanceof Error ? err.message : String(err)} — auto combos disabled` + log.warn( + `/api/combos/auto fetch failed: ${err instanceof Error ? err.message : String(err)} — auto combos disabled` ); return []; } finally { @@ -2935,10 +2950,7 @@ export function passesModelAllowlist( * filter is set, all combos pass. Combos with zero resolvable members pass * (mirrors `isUsableCombo` semantics). */ -export function passesComboAllowlist( - combo: OmniRouteRawCombo, - visible?: ModelListFilter -): boolean { +export function passesComboAllowlist(combo: OmniRouteRawCombo, visible?: ModelListFilter): boolean { if (!visible) return true; const steps = Array.isArray(combo.models) ? combo.models : []; if (steps.length === 0) return true; @@ -3130,9 +3142,15 @@ export function createOmniRouteProviderHook( providersFetcher?: OmniRouteProvidersFetcher; now?: () => number; cache?: OmniRouteFetchCache; + logger?: _Logger; } = {} ): ProviderHook { const resolved = resolveOmniRoutePluginOptions(opts); + const logger = + deps.logger ?? + createLogger( + resolved.features?.startupDebug ? "debug" : (resolved.features?.logLevel ?? "warn") + ); const fetcher = deps.fetcher ?? defaultOmniRouteModelsFetcher; // T-05: combo discovery merges `/api/combos` entries into the same map as // `/v1/models`. Default fetcher is declared further down the file; the @@ -3206,8 +3224,8 @@ export function createOmniRouteProviderHook( : undefined) ?? ""; if (!baseURL) { - console.warn( - `[omniroute-plugin] provider.models(${resolved.providerId}): ` + + logger.error( + `provider.models(${resolved.providerId}): ` + `no baseURL resolvable — checked plugin opts, auth.json, and provider config. ` + `Set baseURL in opencode.json plugin options or run \`opencode connect ${resolved.providerId}\` with a baseURL.` ); @@ -3238,8 +3256,8 @@ export function createOmniRouteProviderHook( rawModels = await fetcher(baseURL, apiKey, 10_000); // T-05: combos fetch is best-effort, gated by features.combos. - // Soft-fail on any error: emit a console.warn and fall back to a - // models-only catalog. Rationale: /api/combos requires a + // Soft-fail on any error: emit a warn-level diagnostic and fall back + // to a models-only catalog. Rationale: /api/combos requires a // management-scoped key and OmniRoute may not have any combos // provisioned. Hard-failing when combos are optional would // silently hide the whole provider from OC's picker. @@ -3248,10 +3266,7 @@ export function createOmniRouteProviderHook( try { rawCombos = await combosFetcher(baseURL, managementReadToken, 10_000); } catch (err) { - console.warn( - "[omniroute-plugin] combos fetch failed, falling back to models-only catalog", - err - ); + logger.warn("combos fetch failed, falling back to models-only catalog", err); } } @@ -3261,7 +3276,7 @@ export function createOmniRouteProviderHook( rawAutoCombos = []; if (wantAutoCombos) { try { - rawAutoCombos = await autoCombosFetcher(baseURL, managementReadToken, 5_000); + rawAutoCombos = await autoCombosFetcher(baseURL, managementReadToken, 5_000, logger); } catch { // Already handled inside the default fetcher — this catch // is belt-and-suspenders for injected stubs. @@ -3275,10 +3290,7 @@ export function createOmniRouteProviderHook( try { rawEnrichment = await enrichmentFetcher(baseURL, managementReadToken, 10_000); } catch (err) { - console.warn( - "[omniroute-plugin] enrichment fetch failed, falling back to raw ids", - err - ); + logger.warn("enrichment fetch failed, falling back to raw ids", err); } } @@ -3293,7 +3305,7 @@ export function createOmniRouteProviderHook( 10_000 ); } catch (err) { - console.warn("[omniroute-plugin] compression-metadata fetch failed", err); + logger.warn("compression-metadata fetch failed", err); } } @@ -3307,8 +3319,8 @@ export function createOmniRouteProviderHook( try { rawConnections = await providersFetcher(baseURL, managementReadToken, 10_000); } catch (err) { - console.warn( - "[omniroute-plugin] /api/providers fetch failed; usableOnly filter disabled for this refresh", + logger.warn( + "/api/providers fetch failed; usableOnly filter disabled for this refresh", err ); } @@ -3327,8 +3339,9 @@ export function createOmniRouteProviderHook( // Debug breadcrumb: surface fetch result so operators can confirm // the dynamic pipeline fired and how much catalog OmniRoute returned. // Emitted once per cache miss (TTL refresh) — quiet on cache hits. - console.warn( - `[omniroute-plugin] catalog refreshed for providerId=${resolved.providerId} baseURL=${baseURL}: ` + + // Info-level: hidden at the default `warn` level (see #8982). + logger.info( + `catalog refreshed for providerId=${resolved.providerId} baseURL=${baseURL}: ` + `${rawModels.length} models + ${rawCombos.length} combos + ` + `${rawEnrichment.size} enrichment entries + ` + `${rawCompressionCombos.length} compression combos + ` + @@ -3608,9 +3621,7 @@ export function createOmniRouteProviderHook( const dedupeKey = `${cacheKey}::${comboKey}`; if (!collisionWarned.has(dedupeKey)) { collisionWarned.add(dedupeKey); - console.warn( - `[omniroute-plugin] combo key "${comboKey}" collides with a model id; combo wins.` - ); + logger.warn(`combo key "${comboKey}" collides with a model id; combo wins.`); } } } @@ -3628,8 +3639,8 @@ export function createOmniRouteProviderHook( } if (pending.length > 0) { - console.warn( - `[omniroute-plugin] ${pending.length} combo(s) could not resolve all nested combo-refs after ${MAX_COMBO_PASSES} passes; they will advertise context=0 to avoid over-claiming.` + logger.warn( + `${pending.length} combo(s) could not resolve all nested combo-refs after ${MAX_COMBO_PASSES} passes; they will advertise context=0 to avoid over-claiming.` ); } @@ -4273,8 +4284,10 @@ export function buildStaticProviderEntry( enrichment?: OmniRouteEnrichmentMap, compressionCombos?: OmniRouteCompressionCombo[], connections?: OmniRouteProviderConnection[], - rawAutoCombos?: OmniRouteRawAutoCombo[] + rawAutoCombos?: OmniRouteRawAutoCombo[], + logger?: OmniRouteLoggerSink ): OmniRouteStaticProviderEntry { + const log = logger ?? _logger; const models: Record = {}; const rawModelKeys = new Set(); @@ -4652,8 +4665,8 @@ export function buildStaticProviderEntry( } if (pendingStatic.length > 0) { - console.warn( - `[omniroute-plugin] ${pendingStatic.length} combo(s) in the static catalog could not resolve all nested combo-refs after ${MAX_STATIC_COMBO_PASSES} passes; they will be omitted.` + log.warn( + `${pendingStatic.length} combo(s) in the static catalog could not resolve all nested combo-refs after ${MAX_STATIC_COMBO_PASSES} passes; they will be omitted.` ); } @@ -4674,9 +4687,7 @@ export function buildStaticProviderEntry( const isExpectedRawTwin = autoCombo.id === key && rawModelKeys.has(key); if (!isExpectedRawTwin && !reportedCollisions.has(key)) { reportedCollisions.add(key); - console.warn( - `[omniroute-plugin] auto combo key "${key}" collides with an existing model; auto combo wins.` - ); + log.warn(`auto combo key "${key}" collides with an existing model; auto combo wins.`); } } models[key] = entry; @@ -5347,7 +5358,8 @@ export function createOmniRouteConfigHook( warmSnapshot = snapshotResult; // Log snapshot age (accept any age — instant beats empty). const age = (snapshotResult as { writtenAt?: number }).writtenAt; - const ageLabel = typeof age === "number" ? `${Math.round((Date.now() - age) / 3_600_000)}h` : "unknown"; + const ageLabel = + typeof age === "number" ? `${Math.round((Date.now() - age) / 3_600_000)}h` : "unknown"; logAt( "warn", `config shim: warm startup from disk snapshot (${snapshotResult.rawModels.length} models, age ${ageLabel})` @@ -5399,7 +5411,12 @@ export function createOmniRouteConfigHook( const doAutoCombos = async (): Promise => { if (!wantAutoCombos) return; try { - localRawAutoCombos = await autoCombosFetcher(baseURL, managementReadToken, 5_000); + localRawAutoCombos = await autoCombosFetcher( + baseURL, + managementReadToken, + 5_000, + logger + ); } catch { // Already handled inside the default fetcher } @@ -5420,7 +5437,11 @@ export function createOmniRouteConfigHook( const doCompression = async (): Promise => { if (!wantCompressionMeta) return; try { - localRawCompressionCombos = await compressionMetaFetcher(baseURL, managementReadToken, 10_000); + localRawCompressionCombos = await compressionMetaFetcher( + baseURL, + managementReadToken, + 10_000 + ); } catch (err) { logAt( "error", @@ -5533,7 +5554,8 @@ export function createOmniRouteConfigHook( localRawEnrichment, localRawCompressionCombos, localRawConnections, - localRawAutoCombos + localRawAutoCombos, + logger ); const inputWithProvider2 = input as { provider?: Record }; if (inputWithProvider2.provider) { @@ -5623,7 +5645,8 @@ export function createOmniRouteConfigHook( rawEnrichment, rawCompressionCombos, rawConnections, - rawAutoCombos + rawAutoCombos, + logger ); // Mutate the input.provider map. The Config type declares diff --git a/@omniroute/opencode-plugin/tests/log-level.test.ts b/@omniroute/opencode-plugin/tests/log-level.test.ts index 7e73cb4eb82..58566aba6d0 100644 --- a/@omniroute/opencode-plugin/tests/log-level.test.ts +++ b/@omniroute/opencode-plugin/tests/log-level.test.ts @@ -5,8 +5,14 @@ import { join } from "node:path"; import test from "node:test"; import type { Config } from "@opencode-ai/plugin"; -import { createOmniRouteConfigHook, OmniRoutePlugin } from "../src/index.js"; -import { getLogLevel, logger, setLogLevel, type LogLevel } from "../src/logger.js"; +import { + createOmniRouteConfigHook, + createOmniRouteProviderHook, + defaultOmniRouteAutoCombosFetcher, + OmniRoutePlugin, + type OmniRouteRawModelEntry, +} from "../src/index.js"; +import { createLogger, getLogLevel, logger, setLogLevel, type LogLevel } from "../src/logger.js"; type ConsoleMethod = "error" | "info" | "log" | "warn"; type ConsoleEntries = Record; @@ -216,3 +222,105 @@ test("logger error output remains visible at error level", async () => { setLogLevel(previousLevel); } }); + +const MINIMAL_MODELS: OmniRouteRawModelEntry[] = [ + { + id: "claude-primary", + object: "model", + owned_by: "combo", + capabilities: { tool_calling: true, reasoning: true, vision: true, thinking: true }, + context_length: 200000, + max_output_tokens: 64000, + input_modalities: ["text", "image"], + output_modalities: ["text"], + }, +]; + +function providerHookWithLevel(level: LogLevel, baseURL?: string) { + return createOmniRouteProviderHook( + { + baseURL, + features: { autoCombos: false, enrichment: false, logLevel: level }, + }, + { + fetcher: async () => MINIMAL_MODELS, + combosFetcher: async () => { + throw new Error("combos boom"); + }, + } + ); +} + +test("logLevel error suppresses provider.models() fallback warnings and the catalog-refresh breadcrumb", async () => { + const hook = providerHookWithLevel("error", "https://or.example.com/v1"); + const lines = rendered( + await captureConsole(async () => { + await hook.models!({} as never, { auth: { type: "api", key: "sk-x" } as never }); + }) + ); + + assert.equal(lines.filter((line) => line.includes("combos fetch failed")).length, 0); + assert.equal(lines.filter((line) => line.includes("catalog refreshed")).length, 0); +}); + +test("logLevel debug preserves the provider.models() catalog-refresh breadcrumb", async () => { + const hook = providerHookWithLevel("debug", "https://or.example.com/v1"); + const lines = rendered( + await captureConsole(async () => { + await hook.models!({} as never, { auth: { type: "api", key: "sk-x" } as never }); + }) + ); + + assert.ok( + lines.some((line) => line.includes("catalog refreshed")), + "catalog-refresh breadcrumb emitted at debug level" + ); +}); + +test("no baseURL resolvable stays visible at error level", async () => { + const hook = providerHookWithLevel("error"); + const lines = rendered( + await captureConsole(async () => { + await hook.models!({} as never, { auth: { type: "api", key: "sk-x" } as never }); + }) + ); + + assert.ok( + lines.some((line) => line.includes("no baseURL resolvable")), + "genuine misconfiguration error remains visible at error level" + ); +}); + +test("default auto-combos fetcher 404 warning respects the threaded logger level", async () => { + const originalFetch = globalThis.fetch; + (globalThis as { fetch: unknown }).fetch = (async () => ({ + status: 404, + ok: false, + })) as typeof fetch; + try { + const silent = await captureConsole(async () => { + await defaultOmniRouteAutoCombosFetcher( + "https://or.example.com/v1", + "sk-x", + 5_000, + createLogger("error") + ); + }); + assert.equal(rendered(silent).length, 0, "404 warning suppressed at error level"); + + const loud = await captureConsole(async () => { + await defaultOmniRouteAutoCombosFetcher( + "https://or.example.com/v1", + "sk-x", + 5_000, + createLogger("warn") + ); + }); + assert.ok( + rendered(loud).some((line) => line.includes("/api/combos/auto not available")), + "404 warning emitted at warn level" + ); + } finally { + globalThis.fetch = originalFetch; + } +}); diff --git a/README.md b/README.md index 0d5ea9f25fd..390acbe4724 100644 --- a/README.md +++ b/README.md @@ -1497,7 +1497,7 @@ OmniRoute stands on the shoulders of giants. It started as a fork of **[9router] - + diff --git a/changelog.d/features/10909-free-provider-rankings-reliability.md b/changelog.d/features/10909-free-provider-rankings-reliability.md new file mode 100644 index 00000000000..e5377885efc --- /dev/null +++ b/changelog.d/features/10909-free-provider-rankings-reliability.md @@ -0,0 +1 @@ +- **feat(rankings):** free provider rankings now expose a `reliability` field (raw `testStatus`/`rateLimitedUntil` per connection plus a `healthy`/`degraded`/`down` state, reusing the `ProviderHealthState` vocabulary of the provider health matrix) when the configured/available filters are active — derived from already-loaded data, without touching the ranking order ([#10909](https://github.com/diegosouzapw/OmniRoute/pull/10909)) diff --git a/changelog.d/features/10920-egress-ip-lock.md b/changelog.d/features/10920-egress-ip-lock.md new file mode 100644 index 00000000000..af7308b62f2 --- /dev/null +++ b/changelog.d/features/10920-egress-ip-lock.md @@ -0,0 +1,8 @@ +- `feat(resilience)`: when an allowlisted provider (opencode family) answers + 429 classified `quota_exhausted` or `rate_limit_exceeded` and its free-tier + quota is bucketed by egress IP (#9611), every connection of that family + sharing the IP is cooled down together before the rotation tries them — one + guaranteed-failed upstream call per episode instead of N, on the combo path + as well. For the allowlisted family a 429 now cools the connection instead + of locking a single model. Exclusive allowlist, never terminal, best-effort + when the egress IP is unknown (#10920). diff --git a/changelog.d/features/10926-rankings-usage-reliability.md b/changelog.d/features/10926-rankings-usage-reliability.md new file mode 100644 index 00000000000..a79b92b4cf6 --- /dev/null +++ b/changelog.d/features/10926-rankings-usage-reliability.md @@ -0,0 +1 @@ +- **feat(rankings):** free provider rankings can now report what each provider actually served — `reliability.usage` (requests, successes, success rate over a window) behind the opt-in `withUsage`/`usageRange` query parameters, so a provider that answers every call with an error is no longer described as healthy ([#10926](https://github.com/diegosouzapw/OmniRoute/pull/10926)) diff --git a/changelog.d/features/command-code-reasoning-efforts.md b/changelog.d/features/command-code-reasoning-efforts.md new file mode 100644 index 00000000000..3e9b172204a --- /dev/null +++ b/changelog.d/features/command-code-reasoning-efforts.md @@ -0,0 +1 @@ +- feat(command-code): advertise low/medium/high/xhigh/max reasoning-effort suffixes for reasoning-capable models in the catalog and Combo Builder, with request-time resolution to reasoning_effort diff --git a/changelog.d/features/m365-copilot-tool-calls.md b/changelog.d/features/m365-copilot-tool-calls.md new file mode 100644 index 00000000000..bfafe08033e --- /dev/null +++ b/changelog.d/features/m365-copilot-tool-calls.md @@ -0,0 +1 @@ +- **feat(providers):** copilot-m365-web now supports OpenAI tool calling — a router planning turn asks the substrate model (as a tool-selection assistant emitting `CALL_TOOL: name({...})` / `NO_TOOL_NEEDED` text, which bypasses its plugin-registry refusal) and validated decisions surface as `tool_calls` with `finish_reason: "tool_calls"` in both stream and non-stream modes; also flattens the full message history (assistant `tool_calls` + compacted tool results) so multi-turn agent loops keep context, replies to SignalR `type:6` keepalives, surfaces `type:3` error frames instead of a silent empty `stop`, and suppresses `writeAtCursor` text from tool-progress frames diff --git a/changelog.d/features/opencode-go-muse-spark-efforts.md b/changelog.d/features/opencode-go-muse-spark-efforts.md new file mode 100644 index 00000000000..25da8482a93 --- /dev/null +++ b/changelog.d/features/opencode-go-muse-spark-efforts.md @@ -0,0 +1 @@ +- feat(opencode-go): expose Muse Spark 1.2 Contributor reasoning-effort aliases (minimal/low/medium/high/xhigh) in the Combo Builder diff --git a/changelog.d/features/per-connection-upstream-timeout.md b/changelog.d/features/per-connection-upstream-timeout.md new file mode 100644 index 00000000000..a5987ed485b --- /dev/null +++ b/changelog.d/features/per-connection-upstream-timeout.md @@ -0,0 +1 @@ +- **feat(providers):** restore the operator-owned upstream timeout tier per connection via `providerSpecificData.timeoutMs` (preempts the maintainer-only model/provider registry tiers and the global `FETCH_TIMEOUT_MS`), and make the combo per-target timeout ceiling follow the selected connection \ No newline at end of file diff --git a/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md b/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md new file mode 100644 index 00000000000..f204684baa2 --- /dev/null +++ b/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md @@ -0,0 +1 @@ +- **fix(executors):** the Meta AI (muse-spark-web) WebSocket send-message timeout now reports the socket's `readyState` at the moment it fires, so a "Meta AI WS timed out" failure can be told apart as either the connection never opening (`readyState=0`) or opening successfully and then going silent (`readyState=1`) — the exact ambiguity that made #10727 undiagnosable from logs alone (#10727). diff --git a/changelog.d/fixes/10798-respect-log-level-provider-catalog.md b/changelog.d/fixes/10798-respect-log-level-provider-catalog.md new file mode 100644 index 00000000000..3a11aab9093 --- /dev/null +++ b/changelog.d/fixes/10798-respect-log-level-provider-catalog.md @@ -0,0 +1 @@ +- **fix(opencode-plugin):** respect log level in provider.models() catalog path so debug/info/warn messages are suppressed when `features.logLevel` is set to `"error"` ([#10798](https://github.com/diegosouzapw/OmniRoute/pull/10798)) — thanks @tientien17 diff --git a/changelog.d/fixes/10854-skills-marketplace-owner.md b/changelog.d/fixes/10854-skills-marketplace-owner.md new file mode 100644 index 00000000000..e80a109f77f --- /dev/null +++ b/changelog.d/fixes/10854-skills-marketplace-owner.md @@ -0,0 +1 @@ +- **fix(skills):** Marketplace-installed skills are available to API-key-scoped requests, including existing SkillsMP and skills.sh installs ([#10854](https://github.com/diegosouzapw/OmniRoute/pull/10854)) — thanks @kriptoburak diff --git a/changelog.d/fixes/10903-loopback-gate-memory-success.md b/changelog.d/fixes/10903-loopback-gate-memory-success.md new file mode 100644 index 00000000000..25720ba1a89 --- /dev/null +++ b/changelog.d/fixes/10903-loopback-gate-memory-success.md @@ -0,0 +1 @@ +- **fix(providers):** the loopback readiness gate no longer memorizes a failed probe — the next caller after 30s starts a fresh probe, and a readiness failure is logged once per probe instead of once per caller ([#10903](https://github.com/diegosouzapw/OmniRoute/pull/10903)) diff --git a/changelog.d/fixes/10935-cloudflare-relay-path-guard.md b/changelog.d/fixes/10935-cloudflare-relay-path-guard.md new file mode 100644 index 00000000000..0799cdc52da --- /dev/null +++ b/changelog.d/fixes/10935-cloudflare-relay-path-guard.md @@ -0,0 +1 @@ +- **fix(relay):** the Cloudflare proxy-relay worker now resolves `x-relay-path` through the shared `resolveRelayTarget()` guard instead of concatenating it onto the validated target. PR #4643 and its follow-up applied that guard to the Deno and Vercel workers; the Cloudflare generator, ported separately from upstream `decolua/9router` PR #1360, kept `fetch(targetBase + relayPath)`. Validating `x-relay-target` and then concatenating is not sufficient — the path re-points the request past the host that was just checked, through userinfo (`/x@evil.com`), a backslash (`\evil.com`), or a protocol-relative path (`//evil.com/x`). The guard is embedded verbatim under a literal `const resolveRelayTarget =` binding so the hardcoded call site still resolves when the SWC-minified standalone build mangles the source function's own name (#6149), and the new regression test pins that property for this worker by renaming the embedded function and re-evaluating the emitted source. The auth check and the private/loopback target guard are unchanged diff --git a/changelog.d/fixes/10936-standalone-server-cjs-esm-scope.md b/changelog.d/fixes/10936-standalone-server-cjs-esm-scope.md new file mode 100644 index 00000000000..824f7df6477 --- /dev/null +++ b/changelog.d/fixes/10936-standalone-server-cjs-esm-scope.md @@ -0,0 +1 @@ +- **fix(build):** the `next` Docker image no longer crashes on boot with `ReferenceError: require is not defined in ES module scope`. The standalone `server.js` is CommonJS, but the `postbuild` colocate step was re-adding `"type":"module"` to the standalone root `package.json` (undoing `assembleStandalone`'s strip) to make its ESM worker bundles load. The `type:module` scope is now written per-worker-directory instead of on the root, so `server.js` stays CommonJS while the workers stay ESM ([#10936](https://github.com/diegosouzapw/OmniRoute/pull/10936), fixes [#10933](https://github.com/diegosouzapw/OmniRoute/issues/10933)) — thanks @arminanton diff --git a/changelog.d/fixes/10941-relay-private-host-guard.md b/changelog.d/fixes/10941-relay-private-host-guard.md new file mode 100644 index 00000000000..53d3aedeece --- /dev/null +++ b/changelog.d/fixes/10941-relay-private-host-guard.md @@ -0,0 +1 @@ +- **fix(relay):** the private/loopback guard the three proxy-relay workers embed no longer misses four host spellings, and now lives in one place instead of three byte-identical inline copies. Driving `new URL(target).hostname` the way the workers do, the previous guard allowed `::` (the unspecified address, which reaches a service bound to the IPv6 loopback), `localhost.` (the FQDN root dot defeated the exact match and every `.localhost`/`.local`/`.internal` suffix rule, so `svc.internal.` slipped too), `::127.0.0.1` (the deprecated IPv4-compatible form — only `::ffff:` was checked), and `feb0::1` (link-local is `fe80::/10`, spanning `fe80`–`febf`, but only the literal `fe80:` spelling matched). The policy moved to `src/lib/proxyRelay/privateHostname.ts` and is embedded verbatim via `Function#toString` under a literal const name, the same mechanism `resolveRelayTarget` already uses for these workers, so a minified standalone build cannot break the call site (#6149). Nothing previously blocked is now allowed. Severity is low — reaching a worker needs the `x-relay-auth` secret and these are edge runtimes where loopback has nothing listening — but the suffix-rule bypass held regardless of runtime diff --git a/changelog.d/fixes/10945-least-used-rotation.md b/changelog.d/fixes/10945-least-used-rotation.md new file mode 100644 index 00000000000..36b23951b25 --- /dev/null +++ b/changelog.d/fixes/10945-least-used-rotation.md @@ -0,0 +1 @@ +- **Account rotation:** make `fallbackStrategy: "least-used"` actually rotate. The strategy sorts on `lastUsedAt` but never wrote it — only the round-robin branch committed — so on a pool where every `last_used_at` was still `NULL` the tie-break fell through to `priority` and returned the same connection on every dispatch ([#10945](https://github.com/diegosouzapw/OmniRoute/issues/10945)). diff --git a/changelog.d/fixes/10947-windows-updater-artifact-name.md b/changelog.d/fixes/10947-windows-updater-artifact-name.md new file mode 100644 index 00000000000..10c90a216da --- /dev/null +++ b/changelog.d/fixes/10947-windows-updater-artifact-name.md @@ -0,0 +1 @@ +- **Desktop auto-update (Windows):** stop the in-app updater 404ing on every release. NSIS used electron-builder's default artifact name, whose spaces GitHub rewrites to `.` on upload while `latest.yml` keeps `-`, so the manifest pointed at `OmniRoute-Setup-X.Y.Z.exe` while the published asset was `OmniRoute.Setup.X.Y.Z.exe`. The name is now set explicitly to the dot form the asset already has, so nothing published changes name ([#10947](https://github.com/diegosouzapw/OmniRoute/issues/10947)). diff --git a/changelog.d/fixes/10953-preserve-provider-effort-tiers.md b/changelog.d/fixes/10953-preserve-provider-effort-tiers.md new file mode 100644 index 00000000000..d509aac61b1 --- /dev/null +++ b/changelog.d/fixes/10953-preserve-provider-effort-tiers.md @@ -0,0 +1 @@ +- **fix(catalog):** preserve provider-declared reasoning effort tiers instead of replacing them with generic defaults ([#10953](https://github.com/diegosouzapw/OmniRoute/pull/10953)) — thanks @xz-dev diff --git a/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md b/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md new file mode 100644 index 00000000000..fd7e61988ba --- /dev/null +++ b/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md @@ -0,0 +1 @@ +- fix(cli): repair hollow externalized package dirs in the nested `/node_modules` bundle location too, not just the top-level one, fixing macOS/Linux Electron `ERR_MODULE_NOT_FOUND` on Turbopack-externalized packages (#7346) diff --git a/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md b/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md new file mode 100644 index 00000000000..e6458879cf0 --- /dev/null +++ b/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md @@ -0,0 +1 @@ +- **Electron packaged smoke test:** add a cold-restart mode (`ELECTRON_SMOKE_COLD_RESTART=1`, wired blocking on the Linux release leg) that relaunches the packaged app against its own persisted `DATA_DIR` and asserts a native SQLite driver was selected instead of the sql.js WASM fallback, closing the regression-test gap flagged in the stale-ABI `better-sqlite3` investigation ([#7592](https://github.com/diegosouzapw/OmniRoute/issues/7592)). diff --git a/changelog.d/fixes/cline-task-id-passthrough.md b/changelog.d/fixes/cline-task-id-passthrough.md new file mode 100644 index 00000000000..a6d2ecec57c --- /dev/null +++ b/changelog.d/fixes/cline-task-id-passthrough.md @@ -0,0 +1 @@ +- **fix(cline):** Preserve client-supplied Cline task IDs and omit the header when clients provide none, preventing request-scoped proxy IDs from being reported as tasks. diff --git a/changelog.d/fixes/combo-sticky-pin-clear-on-disable.md b/changelog.d/fixes/combo-sticky-pin-clear-on-disable.md new file mode 100644 index 00000000000..17abffb7f7b --- /dev/null +++ b/changelog.d/fixes/combo-sticky-pin-clear-on-disable.md @@ -0,0 +1 @@ +- fix(combo): evict in-memory session-stickiness bindings when a combo disables stickiness, so stale pins stop overriding the declared priority order until TTL/restart diff --git a/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md b/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md new file mode 100644 index 00000000000..b26e1c2086d --- /dev/null +++ b/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md @@ -0,0 +1 @@ +- **fix(models):** a model synced from a provider's own `/models` discovery is now enforced at its real context window immediately, instead of waiting up to 24h for the Feature 5004 reconciler's next tick. The request-time token-limit chain resolves the window from `auto:discovery` overrides, which previously were only written at startup and on a 24h interval — so any model synced mid-cycle (models.dev not indexing it yet, no static registry entry) fell through to the provider's static `defaultContextLength` (128K for OpenRouter) while `/v1/models` simultaneously advertised the real window from the same discovery data. Measured: `openrouter/stealth/ox-alpha` advertised `context_length: 1048576` but rejected requests over 128K with `context_length_exceeded` for a full day after its sync. The reconcile now also runs opportunistically (debounced, fire-and-forget) right after a synced catalog write changes. Companion fix: discovery now captures the vendor-declared `reasoning.default_effort` (e.g. OpenRouter `stealth/ox-alpha` declares `max`, normalized to `xhigh`) as `defaultThinkingEffort`, and the OpenAI dispatch path injects it when a request carries no reasoning field of any shape — the lowest-priority default behind a `-{effort}` suffix alias and a static `ModelSpec.defaultReasoningEffort` — so a reasoning model that returns an empty response without an explicit effort gets the vendor default instead of `upstream_empty_response`. diff --git a/changelog.d/maintenance/10859-filesize-baseline-fix.md b/changelog.d/maintenance/10859-filesize-baseline-fix.md new file mode 100644 index 00000000000..aed0a3fab55 --- /dev/null +++ b/changelog.d/maintenance/10859-filesize-baseline-fix.md @@ -0,0 +1 @@ +- fix(quality): rebaseline file-size for #10859's own modelCapabilities.ts/commandCode.ts growth (missed at merge time) diff --git a/changelog.d/maintenance/10889-feature-flag-count-fix.md b/changelog.d/maintenance/10889-feature-flag-count-fix.md new file mode 100644 index 00000000000..fd932219871 --- /dev/null +++ b/changelog.d/maintenance/10889-feature-flag-count-fix.md @@ -0,0 +1 @@ +- fix(quality): bump EXPECTED_FEATURE_FLAG_COUNT to 52 for #10889's own new flag (missed at merge time) diff --git a/changelog.d/maintenance/regen-translate-path-golden-freebuff.md b/changelog.d/maintenance/regen-translate-path-golden-freebuff.md new file mode 100644 index 00000000000..7822df2283a --- /dev/null +++ b/changelog.d/maintenance/regen-translate-path-golden-freebuff.md @@ -0,0 +1 @@ +- chore(test): regenerate the provider/translate-path golden snapshot to reflect freebuff (#10531), fixing a base-red left by that merge (freebuff/freeinference key ordering only, no value changes). diff --git a/changelog.d/maintenance/release-v3850-basereds-eslint-deadcode-vitest-20260819.md b/changelog.d/maintenance/release-v3850-basereds-eslint-deadcode-vitest-20260819.md new file mode 100644 index 00000000000..905d338575c --- /dev/null +++ b/changelog.d/maintenance/release-v3850-basereds-eslint-deadcode-vitest-20260819.md @@ -0,0 +1,20 @@ +- **fix(ci):** drain three more base-reds on `release/v3.8.50` (#9985). ESLint was reporting + 219 errors locally (vs. 25 in the last CI run) — all from `react-hooks/set-state-in-effect`, + `react-hooks/preserve-manual-memoization`, `react-hooks/immutability`, + `react-hooks/static-components`, `react-hooks/refs` and `react-hooks/purity`, six React + Compiler lint rules that `eslint-plugin-react-hooks` v7 turns on by default and that were + never frozen in `config/quality/eslint-suppressions.json` after the dependency bump. Froze + the pre-existing violations for those six rules via ESLint's native + `--suppress-rule`/`--suppressions-location` mechanism (the same pattern already used for + `@next/next/no-location-assign-relative-destination`) — no application code changed, no rule + disabled, only genuinely-new violations stay blocking. `check:dead-code` was at 418 against a + 415 baseline: removed the unused `src/lib/quota/providerCapabilities.ts` file and the unused + `ProviderQuotaMonitor` interface in `providerQuotaTelemetry.ts` (both dead since PR #10148, + 2026-08-18, confirmed via `grep`/knip cross-reference), landing at 416; the residual +1 could + not be attributed to a single recent commit after checking every dead-list entry touched + since the 2026-08-14 baseline measurement, so it is rebaselined with the investigation + recorded in `quality-baseline.json`. `tests/unit/autoCombo/tieredRotation.test.ts`'s + "rotates across all 43 Cerebras connection IDs" case was hitting vitest's 5000ms default + timeout on a 200-iteration synchronous `selectProvider()` loop under shared-devbox + contention (load average 40-60+ observed) — widened its explicit timeout to 20000ms; the + assertion itself is unchanged. diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index 40f74e61f94..8904dd3f263 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -291,6 +291,109 @@ "src/app/(dashboard)/dashboard/HomePageClient.tsx": { "react-hooks/exhaustive-deps": { "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/a2a/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/acp-agents/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/activity/ActivityFeedClient.tsx": { + "react-hooks/purity": { + "count": 1 + }, + "react-hooks/refs": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/analytics/CacheHealthTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/analytics/ProviderUtilizationTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/analytics/RouteExplainabilityTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": { + "react-hooks/immutability": { + "count": 4 + }, + "react-hooks/preserve-manual-memoization": { + "count": 2 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/audit/A2aAuditTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/audit/ComplianceTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/audit/McpAuditTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/batch/components/wizard/CostEstimateStep.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/batch/components/wizard/JsonlValidationStep.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/batch/files/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cache/components/CacheEntriesTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cache/components/ReasoningCacheTab.tsx": { + "react-hooks/purity": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cache/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 } }, "src/app/(dashboard)/dashboard/cli-agents/CliAgentsPageClient.tsx": { @@ -301,6 +404,119 @@ "src/app/(dashboard)/dashboard/cli-code/components/AntigravityToolCard.tsx": { "react-hooks/exhaustive-deps": { "count": 1 + }, + "react-hooks/immutability": { + "count": 3 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/ClaudeClassifierCompatToggle.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/ClaudeToolCard.tsx": { + "react-hooks/immutability": { + "count": 3 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/CliProfileAutoSyncToggles.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/ClineToolCard.tsx": { + "react-hooks/immutability": { + "count": 3 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/CliproxyapiToolCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/CodexToolCard.tsx": { + "react-hooks/immutability": { + "count": 4 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/DroidToolCard.tsx": { + "react-hooks/immutability": { + "count": 3 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/HermesAgentToolCard.tsx": { + "react-hooks/immutability": { + "count": 1 + }, + "react-hooks/purity": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/KiloToolCard.tsx": { + "react-hooks/immutability": { + "count": 3 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/cli-code/components/OpenClawToolCard.tsx": { + "react-hooks/immutability": { + "count": 3 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/combos/ComboControlCenterClient.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/combos/page.tsx": { + "react-hooks/immutability": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 7 + } + }, + "src/app/(dashboard)/dashboard/conductor/ConductorPageClient.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/conductor/FaroChat.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/conversations/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/costs/components/ApiKeyUsageLimitCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 } }, "src/app/(dashboard)/dashboard/costs/costExplorerUtils.ts": { @@ -308,11 +524,258 @@ "count": 1 } }, + "src/app/(dashboard)/dashboard/costs/quota-share/components/PoolWizard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 3 + } + }, + "src/app/(dashboard)/dashboard/costs/quota-share/hooks/usePoolUsage.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/costs/quota-share/hooks/usePools.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/costs/useApiKeyUsageLimits.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/discovery/DiscoveryPageClient.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.tsx": { + "react-hooks/immutability": { + "count": 3 + } + }, + "src/app/(dashboard)/dashboard/endpoint/components/A2ADashboard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/endpoint/components/MCPDashboard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/endpoint/components/NotionSourceCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/endpoint/components/ObsidianSourceCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/free-provider-rankings/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/health/ProviderHealthAutopilotCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/health/ProviderHealthMatrixCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/health/TelemetryCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/mcp/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/memory/components/EditMemoryModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/memory/components/QdrantConfigCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/memory/components/tabs/MemoriesTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/memory/hooks/useEngineStatus.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/memory/hooks/useMemorySettings.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, "src/app/(dashboard)/dashboard/playground/components/tabs/ApiTab.tsx": { "react-hooks/exhaustive-deps": { "count": 1 } }, + "src/app/(dashboard)/dashboard/plugins/[name]/config/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/plugins/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/provider-stats/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + }, + "react-hooks/static-components": { + "count": 7 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/CustomModelsSection.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/ModelCompatPopover.tsx": { + "react-hooks/refs": { + "count": 4 + }, + "react-hooks/set-state-in-effect": { + "count": 3 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/ProviderCcAliasSection.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/ProviderInterceptionSection.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/ProviderParamFilterSection.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditCompatibleNodeModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderSettings.ts": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/providers/components/AddCompatibleProviderModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/hooks/useProviderModels.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/hooks/useProviderUrlFilters.ts": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/providers/hooks/useRiskAcknowledged.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/providers/services/components/DarioAccountPanel.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/providers/services/components/NinerouterModelList.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/radar/RadarCatalogTable.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/radar/intel/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/radar/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/radar/setup/page.tsx": { + "react-hooks/preserve-manual-memoization": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/relay/RelayProxyClient.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/resilience/connections/components/ResilienceConnectionsClient.tsx": { + "react-hooks/purity": { + "count": 1 + }, + "react-hooks/refs": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/runtime/components/ModelCooldownsCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/AccessTokensTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, "src/app/(dashboard)/dashboard/settings/components/AppearanceTab.tsx": { "@next/next/no-img-element": { "count": 4 @@ -321,16 +784,75 @@ "src/app/(dashboard)/dashboard/settings/components/AuthzSection.tsx": { "no-restricted-syntax": { "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/FallbackChainsEditor.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/IPFilterSection.tsx": { + "react-hooks/immutability": { + "count": 1 } }, "src/app/(dashboard)/dashboard/settings/components/MitmProxyTab.tsx": { "@next/next/no-html-link-for-pages": { "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/ModelCapabilityOverridesTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/ModelsDevSyncTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/OneproxyTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/PayloadRulesTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/PoliciesPanel.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/PricingTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 } }, "src/app/(dashboard)/dashboard/settings/components/ProviderAccountRoutingCard.tsx": { "react-hooks/exhaustive-deps": { "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/ResilienceTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 7 + } + }, + "src/app/(dashboard)/dashboard/settings/components/RoutingStrategyCard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 } }, "src/app/(dashboard)/dashboard/settings/components/SessionInfoCard.tsx": { @@ -338,17 +860,93 @@ "count": 1 } }, + "src/app/(dashboard)/dashboard/settings/components/SystemStorageTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/proxy/GlobalConfigTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/settings/components/proxy/SubscriptionTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, "src/app/(dashboard)/dashboard/tools/agent-bridge/components/AgentList.tsx": { "no-restricted-syntax": { "count": 3 } }, - "src/app/(dashboard)/home/page.tsx": { - "no-restricted-imports": { + "src/app/(dashboard)/dashboard/tools/agent-bridge/components/ModelSelectorModal.tsx": { + "react-hooks/set-state-in-effect": { "count": 1 } }, - "src/app/api/auth/login/route.ts": { + "src/app/(dashboard)/dashboard/tools/agent-bridge/components/SetupWizard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/tools/traffic-inspector/components/CustomHostsManager.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/translator/components/MonitorTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/usage/components/EvalsTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/usage/components/ProviderLimits/useCodexResetCreditRedemption.ts": { + "react-hooks/immutability": { + "count": 2 + } + }, + "src/app/(dashboard)/dashboard/usage/components/RateLimitStatus.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/usage/components/SessionsTab.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/webhooks/WebhooksPageClient.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/webhooks/components/AddWebhookWizard.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/dashboard/webhooks/components/WebhookDeliveriesPanel.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/home/ProviderQuotaWidget.tsx": { + "react-hooks/purity": { + "count": 1 + }, + "react-hooks/refs": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/(dashboard)/home/page.tsx": { "no-restricted-imports": { "count": 1 } @@ -813,14 +1411,10 @@ "count": 1 } }, - "src/app/api/settings/require-login/route.ts": { - "no-restricted-imports": { - "count": 1 - } - }, + "src/app/api/settings/route.ts": { "no-restricted-imports": { - "count": 2 + "count": 1 } }, "src/app/api/settings/system-prompt/route.ts": { @@ -998,6 +1592,16 @@ "count": 1 } }, + "src/app/global-error.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/app/status/page.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, "src/domain/costRules.ts": { "no-restricted-syntax": { "count": 1 @@ -1271,14 +1875,50 @@ "count": 1 } }, + "src/shared/components/KiroAuthModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, "src/shared/components/LanguageSelector.tsx": { "@next/next/no-img-element": { "count": 1 } }, + "src/shared/components/ModelSelectModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 4 + } + }, + "src/shared/components/OAuthModal.tsx": { + "react-hooks/set-state-in-effect": { + "count": 4 + } + }, + "src/shared/components/PricingModal.tsx": { + "react-hooks/immutability": { + "count": 1 + } + }, "src/shared/components/ProxyConfigModal.tsx": { "react-hooks/exhaustive-deps": { "count": 1 + }, + "react-hooks/immutability": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/shared/components/ReasoningRoutingRules.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/shared/components/RequestLoggerDetail.sections.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 } }, "src/shared/components/RequestLoggerV2.tsx": { @@ -1289,6 +1929,27 @@ "src/shared/components/Sidebar.tsx": { "@next/next/no-img-element": { "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 2 + } + }, + "src/shared/components/UsageStats.tsx": { + "react-hooks/preserve-manual-memoization": { + "count": 1 + }, + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/shared/components/analytics/useProviderDailyUsage.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, + "src/shared/components/compression/ComboCompressionModeSelect.tsx": { + "react-hooks/set-state-in-effect": { + "count": 1 } }, "src/shared/contracts/quota.ts": { @@ -1296,6 +1957,11 @@ "count": 1 } }, + "src/shared/hooks/cli/useToolBatchStatuses.ts": { + "react-hooks/set-state-in-effect": { + "count": 1 + } + }, "src/shared/services/apiKeyResolver.ts": { "no-restricted-imports": { "count": 1 @@ -1616,11 +2282,7 @@ "count": 11 } }, - "tests/unit/auth-login-route.test.ts": { - "@typescript-eslint/no-explicit-any": { - "count": 1 - } - }, + "tests/unit/auth-ollama-cloud-per-model-403-3027.test.ts": { "@typescript-eslint/no-explicit-any": { "count": 11 @@ -2499,11 +3161,7 @@ "count": 12 } }, - "tests/unit/login-bootstrap-route.test.ts": { - "@typescript-eslint/no-explicit-any": { - "count": 10 - } - }, + "tests/unit/management-password.test.ts": { "@typescript-eslint/no-explicit-any": { "count": 4 @@ -3284,4 +3942,4 @@ "count": 5 } } -} +} \ No newline at end of file diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 72bc0308754..2e7fa58dacd 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -449,12 +449,15 @@ "src/shared/constants/providers/apikey/gateways.ts": 1298, "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1387, "_rebaseline_2026_08_11_v3850_merge_storm_provider_registry": "DRIFT do merge-storm 2026-08-11 (99 PRs mergeados no release/v3.8.50). AddApiKeyModal.tsx (PR #8949 ChatGPT Web provider) e useProviderConnections.ts/ModelSelectModal.tsx (PRs #9011 combo test-all, #9499 image combos) = UI nova legitima acima do cap; gateways.ts = god-file de catalogo de providers que cresceu com PRs #9009/#9421/#9468/#9594 (qualquer split arriscaria corromper o merge de novo — o proprio PR #9421 quebrou o arquivo); bridge.ts (PR #8949) = ponte Chromium vendored; proxyFetch.ts 1207->1220 = drift herdado de merges. Owner autorizou rebaseline com anotacao (2026-08-11).", - "src/lib/modelCapabilities.ts": 1006, + "src/lib/modelCapabilities.ts": 1016, "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 1014, "open-sse/config/imageRegistry.ts": 1034, "src/sse/handlers/chatHelpers.ts": 1019, "src/shared/middleware/chatBodyAdmission.ts": 1005, - "_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file)." + "_rebaseline_2026_08_20_10668_tabitoken_gateway": "#10668 (yawar-aquil) own catalog growth: src/shared/constants/providers/apikey/gateways.ts 1268->1283 (+15, entirely this PR diff -- one new tabitoken gateway entry, data lines only; base moved from 1255 to 1268 via other merges since the PR forked). Not combination drift: reproducible on the PR branch alone, so the WS5.5 release-captain rule does not apply. Extraction is not available -- the file is pure data (own header: \"Pure data; merged by apikey/index.ts via spread\") and already split into 6 family files under apikey/. Same precedent as _rebaseline_2026_08_14_imagetotext_servicekinds (#10275/#10291, gateways.ts 1250->1255, data lines only) and _rebaseline_2026_08_11_v3850_merge_storm_provider_registry (owner-authorized for this same file).", + "open-sse/executors/commandCode.ts": 1038, + "_rebaseline_2026_08_21_10859_vision_bridge_catalog": "#10859 own growth (Vision Bridge fixes #10808/#10809): src/lib/modelCapabilities.ts 1006->1016 (+10, cmd/gpt-5.3-codex* text-only capability resolution) and open-sse/executors/commandCode.ts 988->1023 (+35, Command Code wire-model normalization for bare ids + reasoning field fallback for opencode-routed gateways). Cohesive bug fixes at the existing capability-resolution / executor chokepoints; not extractable mid-fix. Covered by tests/unit/model-capabilities-command-code-codex-textonly-10703.test.ts, tests/unit/command-code-vision.test.ts, tests/unit/opencode-mimo-reasoning-details-nonstream.test.ts. Pushed directly to release (own-session miss: the original rebaseline was made in a throwaway validation worktree and never landed on the PR branch or the release before merge).", + "_rebaseline_2026_08_21_10907_sticky_pin_clear": "#10907 own growth: open-sse/executors/commandCode.ts 1023->1038 (+15, effort-suffix sanitization threading for the sticky-pin-clear fix). Cohesive change at the existing executor chokepoint. Covered by tests/unit/command-code-executor.test.ts." }, "_rebaseline_base_2026_08_10_proxyfetch": "Base-red fix (green-prs sweep, issue #9985): open-sse/utils/proxyFetch.ts 1207 > cap 1000 — new proxied-TLS fetch helper introduced by the Fal reference-image work. Owner-authorized quick rebaseline to green; structural slim tracked for v3.9.0.", "_rebaseline_2026_07_27_v3849_train2": "Merge-train 2 (7 PRs) — owner-approved 2026-07-27. Single entry: chatCore.ts 4955->5006 (#8595, Responses multi-turn image compaction before the context hard-reject). Genuine irreducible growth at the existing compaction chokepoint in handleChatCore — the PR adds a last-resort retry against the concrete budget plus the estimateFinalInputTokens helper, both wired at the pre-existing call site rather than a new branch. Covered by tests/unit/8560-responses-image-compaction.test.ts (4 tests).", @@ -623,4 +626,4 @@ "_rebaseline_2026_08_20_v3850_merge_train_batch1": "Merge-train batch1 (2026-08-19/20, 30 PRs boarded onto release/v3.8.50): gateways.ts 1255->1268 = PR #10722 (Token Kiosk OpenAI-compatible provider gateway catalog entry, +13 declarative lines, same god-file no-split rationale as prior gateways.ts rebaselines); chatHelpers.ts (uncapped, not previously frozen) new 1017 = PR #10797 (relay/bifrost error normalization, +23/-2, own-PR growth, existing file already near cap from accumulated chokepoint wiring per its own rebaseline history above); chatBodyAdmission.ts (uncapped) new 1005 = pre-existing base-red on the pure release tip (1004>1000 before this train boarded anything, no PR in this batch touches this file) — frozen here at its current size, not authorizing further growth. Owner-authorized rebaseline (2026-08-19 merge-prs session).", "_rebaseline_2026_08_20_8338_cursor_image_provider": "PR (reimplementation of #8338, @valvesss): imageRegistry.ts 1019->1033 = new cursor IMAGE_PROVIDERS entry (Cursor plan image generation via Agent CLI), +14 lines of declarative provider metadata. Same god-registry no-split rationale as prior imageRegistry/gateways rebaselines.", "_rebaseline_2026_08_20_imageregistry_1034": "imageRegistry.ts 1033->1034: +1 line drift between #10842 (cursor image provider, froze at 1033) and its actual merged state on release (measured 1034) — trivial rebaseline, not a new feature." -} \ No newline at end of file +} diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index eae9067b4b0..7fa873b960d 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -102,8 +102,9 @@ "_rebaseline_2026_07_28_v3849_release": "75.5 -> 99 (+23.5). Aperto EXIGIDO pelo modo --require-tighten do ratchet: a métrica melhorou de verdade no ciclo v3.8.49. A causa é o workflow assíncrono de tradução, que finalmente alcançou o denominador em EN — as rebaselines anteriores (v3.8.39/.44/.47) foram todas afrouxamentos registrando o atraso das traduções, e agora ele foi pago. O coletor SUBTRAI os placeholders (present - placeholder em scripts/quality/collect-metrics.mjs), então os 317 marcadores __MISSING__ que esta release introduziu para o drift de valor já estão descontados dos 99 — o número é honesto, não inflado por placeholder. Medido pelo collect-metrics do CI no run 30404226939." }, "deadExports": { - "value": 415, + "value": 416, "direction": "down", + "_rebaseline_2026_08_19_v3850_basereds_9985": "415 -> 416. Measured on release/v3.8.50 tip 14a480453 during the #9985 base-red drain. Removed the 2 genuinely-dead symbols traced to a specific recent change (PR #10148, 2026-08-18): the unused src/lib/quota/providerCapabilities.ts file and the unused ProviderQuotaMonitor interface in providerQuotaTelemetry.ts (418 -> 416). The remaining +1 could not be attributed to a single recent commit after checking every dead-list entry touched since the 2026-08-14 baseline measurement (most are pre-existing debt on files edited for unrelated reasons); rebaselining the residual 1 rather than guessing at removals. Structural cleanup stays tracked in #3501.", "_rebaseline_2026_08_09_v3850_post_sweep": "227 -> 230. Measured by npm run check:dead-code on the unmodified release/v3.8.50 tip 382449d593 during the mandatory --full-ci pre-flight. The +3 is inherited cycle drift from the authorized merge sweep; this repair adds no production exports. Rebaseline records the actual tip so ci.yml quality-gate can run, while structural cleanup remains separate debt.", "_rebaseline_2026_07_01_v3843_release": "225->227 (+2). v3.8.43 cycle drift, surfaced in the Quality Ratchet job after eslintWarnings was rebaselined (check:dead-code runs there). 227 = measured by check:dead-code (knip) on the release tip 4635076eb. The 5 CI fixes add 0 dead exports: safeHttpHref in linkify.ts is module-local AND used (called by linkifyText); no new exports; test files are not scanned. Tighten via --update next cycle.", "dedicatedGate": true, @@ -112,7 +113,8 @@ "_rebaseline_2026_06_26_v3837_release": "343->345. v3.8.37 cycle drift surfaced by the release-green pre-flight (the Quality Ratchet does NOT run on PR->release fast-gates, so warnings/complexity accrued unmeasured across this cycle's 76 commits — provider adds DGrid/Pioneer/xAI, headroom proxy lifecycle #4649, ~50 SSE/translator fixes, Engine Combos #5062). Trust-but-verify: this release-finalize working tree touches ONLY CHANGELOG.md, docs/i18n/*/CHANGELOG.md mirrors, and these baselines — 0 production-code change, so all drift is inherited cycle drift (`any` warn-allowed in open-sse/ + tests/). Tighten via --require-tighten next cycle.", "_rebaseline_2026_08_11_v3850_merge_storm": "230 -> 248. Own drift from the 2026-08-11 merge storm (99 PRs into release/v3.8.50 via authorized sweep): new providers/executors/handlers added dead exports that knip cannot see as used. Measured on the base-fix tip (7ca73697b0 + this repair PR). Owner authorized rebaseline (2026-08-11) — structural cleanup remains separate debt.", "_rebaseline_2026_08_13_v3850_knip_bump": "248 -> 409. NOT code-added dead exports: dependabot bump #10043 (2026-08-13) upgraded knip 6.27.0 -> 6.32.x, and the new knip detects 162 MORE genuinely-unused exports (331 vs 169 deadExports) that 6.27 missed. DEAD_FILES unchanged (78). Reproduced identically on the clean release/v3.8.50 tip 266e39d3 with a fresh knip 6.32 node_modules — so every PR is born red on this gate until the tool change is absorbed. Owner authorized rebaseline (2026-08-13, via base-reds PR #10260). Structural cleanup of the 162 newly-surfaced dead exports remains separate debt.", - "_rebaseline_2026_08_14_ocr_imagetotext_series": "OCR/image-to-text series (#10275/#10283/#10287/#10289/#10291): deadExports 409 -> 415. Each PR in the series adds public util/registry exports that are exercised by their unit tests but not yet by a second production caller — normalizeImageBuffer (imageNormalize), MISTRAL_PASSTHROUGH / AZURE_DI_TRANSFORMATION / getOcrTransformation (ocrRegistry), resolveOcrCredentials (v1/ocr route). They are the documented public surface of the new modules and are covered by tests; structural cleanup stays tracked in #3501." + "_rebaseline_2026_08_14_ocr_imagetotext_series": "OCR/image-to-text series (#10275/#10283/#10287/#10289/#10291): deadExports 409 -> 415. Each PR in the series adds public util/registry exports that are exercised by their unit tests but not yet by a second production caller — normalizeImageBuffer (imageNormalize), MISTRAL_PASSTHROUGH / AZURE_DI_TRANSFORMATION / getOcrTransformation (ocrRegistry), resolveOcrCredentials (v1/ocr route). They are the documented public surface of the new modules and are covered by tests; structural cleanup stays tracked in #3501.", + "_rebaseline_2026_08_20_pr_10798": "415 -> 418. Inherited cycle drift from parallel merges into release/v3.8.50 since the 2026-08-14 OCR-series rebaseline (3 more dead exports surfaced by knip 6.32). This PR (#10798, omniroute-plugin log-level fix) adds 0 production exports: it touches @omniroute/opencode-plugin (separate workspace, not scanned), changelog.d/, and scripts/check/check-env-doc-sync.mjs (array entries, not exports). The +3 is NOT from this PR; rebaselined so the gate runs while structural cleanup of the newly-surfaced dead exports remains separate debt." }, "cognitiveComplexity": { "value": 1223, diff --git a/contrib/vps/.env.example b/contrib/vps/.env.example new file mode 100644 index 00000000000..31b5a38feec --- /dev/null +++ b/contrib/vps/.env.example @@ -0,0 +1,21 @@ +# Build this local image from the exact release checkout as documented below, +# or replace it with an immutable published image digest. +OMNIROUTE_IMAGE=omniroute:3.8.50-vps + +# The dashboard is loopback-only by default. Keep this value unless a trusted +# reverse proxy or private overlay network is configured on the same host. +OMNIROUTE_BIND_HOST=127.0.0.1 +OMNIROUTE_PORT=20128 + +# Generate unique values before the first start. Do not commit the resulting .env. +JWT_SECRET= +API_KEY_SECRET= +OMNIROUTE_WS_BRIDGE_SECRET= +INITIAL_PASSWORD= +REQUIRE_API_KEY=true + +# Conservative defaults for a small VPS. Adjust after observing real usage. +OMNIROUTE_MEMORY_LIMIT=1536m +OMNIROUTE_CPUS=1.0 +OMNIROUTE_PIDS_LIMIT=256 +APP_LOG_LEVEL=info diff --git a/contrib/vps/README.md b/contrib/vps/README.md new file mode 100644 index 00000000000..521f468f95d --- /dev/null +++ b/contrib/vps/README.md @@ -0,0 +1,151 @@ +# Headless Linux VPS deployment + +This bundle runs the published OmniRoute server image on a Linux VPS without +the Electron desktop shell. It keeps the dashboard on loopback by default, +does not publish Redis, persists application data, and adds conservative +resource and log limits. + +Use this bundle when the VPS only needs the API and web dashboard. The existing +root-level Compose profiles remain the right choice for local development, +building from source, bundled provider CLIs, or the Playwright/Chromium image. + +## Prerequisites + +- A supported Linux distribution with Docker Engine and Docker Compose v2. +- At least 2 GiB of available RAM for the default limits. The host needs more + headroom if other workloads run beside OmniRoute. +- SSH access for the loopback dashboard tunnel. + +## Install + +Build the headless server image from the exact release checkout. Building it +locally avoids assuming that a matching version tag has already been published +to a container registry: + +```bash +git switch --detach release/v3.8.50 +test "$(node -p "require('./package.json').version")" = "3.8.50" +docker build --target runner-base --tag omniroute:3.8.50-vps . +``` + +Then initialize the deployment from the repository root: + +```bash +cd contrib/vps +cp .env.example .env +chmod 600 .env +``` + +Generate separate values for every secret, then paste them into `.env`: + +```bash +openssl rand -base64 48 # JWT_SECRET +openssl rand -hex 32 # API_KEY_SECRET +openssl rand -base64 48 # OMNIROUTE_WS_BRIDGE_SECRET +openssl rand -base64 24 # INITIAL_PASSWORD +``` + +Do not reuse these values across installations. Keep `REQUIRE_API_KEY=true`. +Keep `OMNIROUTE_IMAGE` on the locally built version tag, or replace it with an +immutable registry digest; do not use the floating `latest` or `next` tags for +unattended production. + +Validate and start the stack: + +```bash +docker compose config --quiet +docker compose up -d +docker compose ps +``` + +The dashboard is intentionally bound to `127.0.0.1`. Reach it through SSH: + +```bash +ssh -L 20128:127.0.0.1:20128 user@your-vps +``` + +Then open `http://127.0.0.1:20128` locally. For a public hostname, put a trusted +reverse proxy on the same host in front of the loopback port and terminate TLS +there. Do not change `OMNIROUTE_BIND_HOST` to `0.0.0.0` merely to make the +dashboard reachable. + +## Verify + +```bash +docker compose ps +curl --fail --silent http://127.0.0.1:20128/healthz +docker compose logs --tail=100 omniroute +``` + +`/healthz` is a lifecycle probe. Use the authenticated monitoring/API routes +for deeper provider validation after the first login. + +## Web-session providers on a VPS + +Consumer web-session providers can enforce IP reputation, TLS fingerprint, or +browser-session binding. A cookie copied on a workstation may therefore fail +from a datacenter VPS even when the Linux container is healthy. In particular, +Grok clearance cookies can be tied to the browser IP, User-Agent, and TLS +fingerprint. Prefer official API credentials for unattended workloads. When a +web-session provider is required, use only credentials from an account you own +and follow that provider's guide; do not weaken TLS verification or bypass an +access challenge. + +## Backup + +Stop writes before copying SQLite data, then archive the named volume: + +```bash +docker compose stop omniroute +mkdir -p backups +docker run --rm \ + -v omniroute-vps_omniroute-data:/data:ro \ + -v "$PWD/backups:/backup" \ + docker.io/library/alpine:3.23 \ + tar -C /data -czf /backup/omniroute-data.tar.gz . +docker compose start omniroute +``` + +Verify the archive before relying on it: + +```bash +tar -tzf backups/omniroute-data.tar.gz >/dev/null +``` + +Store a timestamped copy outside the VPS. The fixed filename above is kept +simple for copy/paste; rename it after each verified backup. + +## Update and rollback + +Before updating, record the currently running immutable digest and take a +verified backup: + +```bash +docker image inspect "$(docker compose images -q omniroute)" \ + --format '{{index .RepoDigests 0}}' +``` + +Build the new local version tag first, or set `OMNIROUTE_IMAGE` in `.env` to a +new immutable registry digest. Pull only when the selected image is remote, +then recreate the application container: + +```bash +# Registry images only: docker compose pull omniroute +docker compose up -d --no-deps omniroute +docker compose ps +curl --fail --silent http://127.0.0.1:20128/healthz +``` + +To roll back the application image, restore the previous value of +`OMNIROUTE_IMAGE` and repeat the applicable `pull` and `up` commands. Restore the data +archive only when a migration changed the persisted data and image rollback +alone is insufficient. Keep the stack stopped while restoring the volume. + +## Remove the stack + +```bash +docker compose down +``` + +This preserves both named volumes. `docker compose down -v` deletes persistent +data and is intentionally not part of the normal uninstall path. diff --git a/contrib/vps/compose.yaml b/contrib/vps/compose.yaml new file mode 100644 index 00000000000..4e7c43ebd64 --- /dev/null +++ b/contrib/vps/compose.yaml @@ -0,0 +1,69 @@ +name: omniroute-vps + +services: + redis: + image: docker.io/library/redis:8.6.5-alpine + restart: unless-stopped + command: ["redis-server", "--save", "60", "1", "--appendonly", "yes", "--loglevel", "warning"] + volumes: + - redis-data:/data + healthcheck: + test: ["CMD", "redis-cli", "ping"] + interval: 10s + timeout: 5s + retries: 3 + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" + + omniroute: + image: ${OMNIROUTE_IMAGE:?Set OMNIROUTE_IMAGE to a versioned tag or digest} + restart: unless-stopped + stop_grace_period: 40s + depends_on: + redis: + condition: service_healthy + env_file: + - .env + environment: + NODE_ENV: production + PORT: "20128" + DASHBOARD_PORT: "20128" + HOSTNAME: 0.0.0.0 + DATA_DIR: /app/data + REDIS_URL: redis://redis:6379 + REQUIRE_API_KEY: ${REQUIRE_API_KEY:-true} + JWT_SECRET: ${JWT_SECRET:?Set JWT_SECRET in .env} + API_KEY_SECRET: ${API_KEY_SECRET:?Set API_KEY_SECRET in .env} + INITIAL_PASSWORD: ${INITIAL_PASSWORD:?Set INITIAL_PASSWORD in .env} + OMNIROUTE_WS_BRIDGE_SECRET: ${OMNIROUTE_WS_BRIDGE_SECRET:?Set OMNIROUTE_WS_BRIDGE_SECRET in .env} + ports: + - "${OMNIROUTE_BIND_HOST:-127.0.0.1}:${OMNIROUTE_PORT:-20128}:20128" + volumes: + - omniroute-data:/app/data + tmpfs: + - /tmp:size=256m,mode=1777 + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + pids_limit: ${OMNIROUTE_PIDS_LIMIT:-256} + mem_limit: ${OMNIROUTE_MEMORY_LIMIT:-1536m} + cpus: ${OMNIROUTE_CPUS:-1.0} + healthcheck: + test: ["CMD", "node", "healthcheck.mjs"] + interval: 30s + timeout: 5s + retries: 3 + start_period: 20s + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" + +volumes: + omniroute-data: + redis-data: diff --git a/docs/architecture/RESILIENCE_GUIDE.md b/docs/architecture/RESILIENCE_GUIDE.md index d92aebb512c..030f65dd414 100644 --- a/docs/architecture/RESILIENCE_GUIDE.md +++ b/docs/architecture/RESILIENCE_GUIDE.md @@ -369,7 +369,7 @@ matching provider classification rule (`agentrouter-model-access-denied` in `open-sse/config/providerErrorRules.ts`: `reason: "auth_error"`, `scope: "model"`, a `6h` declared base cooldown) is consulted by `checkFallbackError` (`open-sse/services/accountFallback.ts`) -*before* the generic apikey-category `FORBIDDEN` early-return, gated on +_before_ the generic apikey-category `FORBIDDEN` early-return, gated on `honorsRuleLockScope(provider)` (#10334 — currently agentrouter-exclusive via the `HONORS_RULE_LOCK_SCOPE_PROVIDERS` allowlist in `providerErrorRules.ts`). The rule's declared 6h cooldown flows through as @@ -378,7 +378,7 @@ per-model-quota lockout path (`lockModelIfPerModelQuota()` / `recordModelLockoutFailure()`, unchanged by #10334 except for the cooldown source): it is clamped down to the operator's `mlSettings.maxCooldownMs` (default `1_800_000ms` / 30min), like every other model lockout, and the -*persisted lockout reason* stays the pre-existing hardcoded `"forbidden"`, +_persisted lockout reason_ stays the pre-existing hardcoded `"forbidden"`, not the rule's `"auth_error"` — only the cooldown duration is honored end-to-end, not the reason string. The connection itself stays active; sibling models on the same connection are unaffected. @@ -413,7 +413,7 @@ never `creditsExhausted` — a defense against a future rule pairing scope `open-sse/services/combo/targetExhaustion.ts`): the same guard marks the connection into the in-memory `exhaustedConnections` set, keyed `${provider}:${connectionId}`. This only skips a remaining SAME-REQUEST - target that *itself already carries that exact `connectionId`* on its own + target that _itself already carries that exact `connectionId`_ on its own target object (`getExhaustedTargetSkipReason()`, `open-sse/services/combo/comboPredicates.ts`, `if (provider && connectionId)` before the `exhaustedConnections` lookup) — a plain @@ -498,6 +498,68 @@ provider is on that allowlist. No changes to `chatCore.ts`, `classifyError`, or combo are needed. +#### Egress-bucketed lock (#10880) + +Providers in `EGRESS_BUCKETED_LOCK_PROVIDERS` (opencode family) are treated +as IP-bucketed upstream (the opencode free tier is IP-bucketed, not +account-bucketed — see #9611): a status-429 classified `quota_exhausted` +**or** `rate_limit_exceeded` cools down every allowlisted-family connection +whose last known egress IP matches the failing connection's, before the +rotation can try them +— avoiding N-1 guaranteed-failed upstream calls (same shape as #10460/#10525). +`rate_limit_exceeded` is included deliberately: on the `markAccountUnavailable` +path the opencode-specific rules never match (no headers/body handed to +`checkFallbackError`, opencode not in `FULL_TEXT_RULE_PROVIDERS`), so a 429 +whose body carries the subscription-quota text ("monthly usage limit +reached") is classified `quota_exhausted` by the quota-text fallback +(`buildSubscriptionQuotaFallback`, `accountFallback.ts`; 1h cooldown) before +the `status_429` rule is ever reached — while a quota-text-free 429 (plain +rate limiting) classifies via the `status_429` rule as `rate_limit_exceeded` +and still cools the IP family down. For an allowlisted provider an IP-bucketed +rate limit is the same signal as an exhausted quota. Honest limits: + +- **Best-effort**: the lock resolves the connection's last known `egress_ip` + from `proxy_logs` (24h window, synchronous, no cache). Cold cache (egress + IP never probed) or no row → the failing connection is still cooled by the + branch (recorded like today), only no sibling is locked. +- **Never terminal**: the cooldown is a renewing quota window + (`testStatus: "unavailable"`); a permanent state is never derived from an + IP-level signal. `disableCooling` connections skip the branch entirely. +- **Lock granularity changes for the allowlisted family**: this is a scope + change, not only a sibling optimization. opencode is a `passthroughModels` + provider, so before this branch a 429 produced a per-MODEL lockout; it now + produces a connection cooldown — including for an operator running a single + connection with no sibling at all. That is the granularity the opencode rule + table already declares correct (`scope: "connection"`, + `providerErrorRules.ts`), never honored so far because opencode is not in + `HONORS_RULE_LOCK_SCOPE_PROVIDERS`. The branch writes the failing + connection's cooldown + `backoffLevel` itself, mirroring the + connection-scoped agentrouter branch, and returns — the per-model block and + the generic path below are never reached. +- **Combo included**: like the agentrouter branch, the scope deliberately + ignores the `persistUnavailableState`/`isCombo` downgrade a combo caller + applies to a 429. A per-model lockout is not a weaker form of this scope, it + is the wrong unit: it says nothing about the exhausted IP, so the combo + rotation would keep burning one guaranteed-failed call per sibling. +- **Sibling safety**: a sibling already terminal (banned/credits_exhausted) + or already in a longer cooldown is never overwritten. +- **Exclusive allowlist**: widening `EGRESS_BUCKETED_LOCK_PROVIDERS` is an + explicit owner decision; no generic wiring (pattern #10334/#10419). The + sibling query binds that same allowlist rather than repeating it as a SQL + literal, so widening it stays a one-line change. +- **Egress IP rotation, both directions**: the lookup window (24h) is far + wider than the egress-IP cache TTL (5 min), so "last known IP" is history, + not current state. If a connection's proxy rotated within the window the + lock may **miss** a genuinely shared IP (the recorded IP is the new, + unexhausted one) — and symmetrically it may **cool a sibling that has since + rotated away** from the exhausted IP. The second case costs that sibling one + cooldown window; both are accepted best-effort limits of a history-based + lookup. +- **Cost**: two bounded scans of `proxy_logs` (window-filtered via + `idx_pl_timestamp`), only at 429 frequency. No new index (migration 134 + YAGNI). Measured on a real-traffic DB copy of moderate size; a + high-throughput instance holds proportionally more rows in the same window. + --- ## Other Resilience Features diff --git a/docs/guides/ELECTRON_GUIDE.md b/docs/guides/ELECTRON_GUIDE.md index bdfcc247896..340e2a2c8df 100644 --- a/docs/guides/ELECTRON_GUIDE.md +++ b/docs/guides/ELECTRON_GUIDE.md @@ -252,7 +252,7 @@ AppImage signing is optional — set `LINUX_GPG_KEY` if signing. Artifacts land in `electron/dist-electron/`: -- `OmniRoute Setup X.Y.Z.exe`, `OmniRoute-X.Y.Z-portable.exe` (Windows) +- `OmniRoute.Setup.X.Y.Z.exe`, `OmniRoute X.Y.Z.exe` (Windows) - `OmniRoute-X.Y.Z-mac.dmg`, `OmniRoute-X.Y.Z-arm64-mac.dmg` (macOS) - `OmniRoute-X.Y.Z.AppImage`, `omniroute-desktop_X.Y.Z_amd64.deb` (Linux) diff --git a/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md b/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md index 7bcf2bb0cfa..1b0495a1b01 100644 --- a/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md +++ b/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md @@ -252,7 +252,7 @@ Podpis AppImage jest opcjonalny — ustaw `LINUX_GPG_KEY`, jeśli podpisujesz. Artefakty lądują w `electron/dist-electron/`: -- `OmniRoute Setup X.Y.Z.exe`, `OmniRoute-X.Y.Z-portable.exe` (Windows) +- `OmniRoute.Setup.X.Y.Z.exe`, `OmniRoute X.Y.Z.exe` (Windows) - `OmniRoute-X.Y.Z-mac.dmg`, `OmniRoute-X.Y.Z-arm64-mac.dmg` (macOS) - `OmniRoute-X.Y.Z.AppImage`, `omniroute-desktop_X.Y.Z_amd64.deb` (Linux) diff --git a/electron/package.json b/electron/package.json index 57789a43189..bdb8b205f5d 100644 --- a/electron/package.json +++ b/electron/package.json @@ -135,6 +135,7 @@ "category": "Utility" }, "nsis": { + "artifactName": "${productName}.Setup.${version}.${ext}", "oneClick": false, "allowToChangeInstallationDirectory": true, "createDesktopShortcut": true, diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index fe30a37325d..8236a4d5d8f 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -439,7 +439,8 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "ovhcloud", modelId: "Qwen3.6-27B", displayName: "Qwen3.6 27B (OVH anonymous)", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "ovhcloud-anon", tos: "ok" }, { provider: "ovhcloud", modelId: "Mistral-Small-3.2-24B-Instruct-2506", displayName: "Mistral Small 3.2 24B (OVH anonymous)", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "ovhcloud-anon", tos: "ok" }, { provider: "ovhcloud", modelId: "Qwen2.5-VL-72B-Instruct", displayName: "Qwen2.5 VL 72B (OVH anonymous)", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "ovhcloud-anon", tos: "ok" }, - { provider: "agnes", modelId: "agnes-2.5-pro", displayName: "Agnes 2.5 Pro", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "agnes-free", tos: "ok" }, + { provider: "agnes", modelId: "agnes-1.5-flash", displayName: "Agnes 1.5 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "agnes-free", tos: "ok" }, + { provider: "agnes", modelId: "agnes-2.0-flash", displayName: "Agnes 2.0 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "agnes-free", tos: "ok" }, { provider: "agnes", modelId: "agnes-2.5-flash", displayName: "Agnes 2.5 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "agnes-free", tos: "ok" }, { provider: "glm", modelId: "glm-4.7-flash", displayName: "GLM-4.7-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, { provider: "glm", modelId: "glm-4.5-flash", displayName: "GLM-4.5-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, diff --git a/open-sse/config/providerErrorRules.ts b/open-sse/config/providerErrorRules.ts index 17d72be598d..c910e3fb4d9 100644 --- a/open-sse/config/providerErrorRules.ts +++ b/open-sse/config/providerErrorRules.ts @@ -53,11 +53,13 @@ export type ProviderErrorRuleMatch = { // every model on the same provider until the 5h window resets. // // Scope note: `scope: "connection"` (not "provider") is correct because the -// upstream quota is per-account, and a single OmniRoute provider entry maps to -// one user account. Multiple OmniRoute connections under the same provider -// name mean the user has multiple upstream accounts — locking at the provider -// level would disable every one of them when only one is exhausted. See -// Issue #2 (Monthly quota exhausted treated as transient 429). +// upstream quota is per egress IP for the free tier (the opencode free tier +// is IP-bucketed, not account-bucketed — see #9611) and per account for paid +// plans; a single OmniRoute provider entry maps to one user account. Multiple +// OmniRoute connections under the same provider name mean the user has +// multiple upstream accounts — locking at the provider level would disable +// every one of them when only one is exhausted. See Issue #2 (Monthly quota +// exhausted treated as transient 429) and #10880 (egress-bucketed cooldown). function buildOpencodeRules(): ProviderErrorRule[] { return [ { @@ -277,6 +279,36 @@ export function honorsRuleLockScope(provider: string | null | undefined): boolea return !!provider && HONORS_RULE_LOCK_SCOPE_PROVIDERS.has(provider.toLowerCase()); } +/** + * Providers whose upstream quota is bucketed by EGRESS IP, not by account — + * the opencode free tier is IP-bucketed, not account-bucketed (see #9611). + * When such a provider answers 429 quota_exhausted or + * rate_limit_exceeded (see the markAccountUnavailable branch comment — the + * real opencode 429 arrives as rate_limit_exceeded on that path), every + * connection egressing through that IP shares the exhausted budget, so the + * lock is applied at egress-IP scope (see markAccountUnavailable / + * applyEgressIpLockout). EXCLUSIVE allowlist by design — same pattern as + * HONORS_RULE_LOCK_SCOPE_PROVIDERS (#10334): a provider must opt in, and any + * widening is an explicit owner decision. + */ +const EGRESS_BUCKETED_LOCK_PROVIDERS = new Set(["opencode", "opencode-go", "opencode-cli"]); + +export function isEgressBucketedLockScope(provider: string | null | undefined): boolean { + return !!provider && EGRESS_BUCKETED_LOCK_PROVIDERS.has(provider.toLowerCase()); +} + +/** + * The same allowlist as a sorted array, for callers that must express it as + * data rather than a predicate (the sibling lookup in `applyEgressIpLockout` + * binds it into a SQL `IN (...)`). Single source of truth on purpose: a + * literal provider list duplicated in a query would silently NOT follow a + * widening of `EGRESS_BUCKETED_LOCK_PROVIDERS`, leaving the opt-in half + * applied. + */ +export function egressBucketedLockProviders(): string[] { + return [...EGRESS_BUCKETED_LOCK_PROVIDERS].sort(); +} + /** * Providers whose rules match on the FULL upstream error text. * checkFallbackError's rule lookup normally passes only the structured diff --git a/open-sse/config/providers/registry/agnes/index.ts b/open-sse/config/providers/registry/agnes/index.ts index 848ad0242c6..2843328f00b 100644 --- a/open-sse/config/providers/registry/agnes/index.ts +++ b/open-sse/config/providers/registry/agnes/index.ts @@ -2,21 +2,28 @@ import type { RegistryEntry } from "../../shared.ts"; export const agnesProvider: RegistryEntry = { id: "agnes", - format: "openai-responses", + format: "openai", executor: "default", - baseUrl: "https://apihub.agnes-ai.com/v1/responses", + baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions", authType: "apikey", authHeader: "bearer", models: [ { - id: "agnes-2.5-pro", - name: "Agnes 2.5 Pro", - contextLength: 1048576, + id: "agnes-1.5-flash", + name: "Agnes 1.5 Flash", + contextLength: 262144, + maxOutputTokens: 65536, + supportsVision: true, + toolCalling: true, + }, + { + id: "agnes-2.0-flash", + name: "Agnes 2.0 Flash", + contextLength: 262144, maxOutputTokens: 65536, supportsReasoning: true, supportsVision: true, toolCalling: true, - interleavedField: "reasoning_content", }, { id: "agnes-2.5-flash", diff --git a/open-sse/config/providers/registry/command-code/index.ts b/open-sse/config/providers/registry/command-code/index.ts index 23a73efe462..77b08ab8ce8 100644 --- a/open-sse/config/providers/registry/command-code/index.ts +++ b/open-sse/config/providers/registry/command-code/index.ts @@ -1,5 +1,7 @@ import type { RegistryEntry } from "../../shared.ts"; +const COMMAND_CODE_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"] as const; + export const command_codeProvider: RegistryEntry = { id: "command-code", alias: "cmd", @@ -20,6 +22,7 @@ export const command_codeProvider: RegistryEntry = { id: "claude-opus-4-7", name: "Claude Opus 4.7 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 200000, maxOutputTokens: 32000, @@ -28,6 +31,7 @@ export const command_codeProvider: RegistryEntry = { id: "claude-opus-4-6", name: "Claude Opus 4.6 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 200000, maxOutputTokens: 32000, @@ -36,6 +40,7 @@ export const command_codeProvider: RegistryEntry = { id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 200000, maxOutputTokens: 16384, @@ -44,6 +49,7 @@ export const command_codeProvider: RegistryEntry = { id: "claude-haiku-4-5-20251001", name: "Claude Haiku 4.5 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 200000, maxOutputTokens: 8192, @@ -52,6 +58,7 @@ export const command_codeProvider: RegistryEntry = { id: "gpt-5.5", name: "GPT-5.5 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 256000, maxOutputTokens: 128000, @@ -60,6 +67,7 @@ export const command_codeProvider: RegistryEntry = { id: "gpt-5.4", name: "GPT-5.4 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 256000, maxOutputTokens: 128000, @@ -68,6 +76,7 @@ export const command_codeProvider: RegistryEntry = { id: "gpt-5.3-codex", name: "GPT-5.3 Codex (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 256000, maxOutputTokens: 128000, @@ -75,7 +84,8 @@ export const command_codeProvider: RegistryEntry = { { id: "gpt-5.4-mini", name: "GPT-5.4 Mini (CC)", - supportsReasoning: false, + supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 256000, maxOutputTokens: 128000, @@ -84,6 +94,7 @@ export const command_codeProvider: RegistryEntry = { id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 1000000, maxOutputTokens: 131072, }, @@ -91,6 +102,7 @@ export const command_codeProvider: RegistryEntry = { id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 1000000, maxOutputTokens: 131072, }, @@ -98,6 +110,7 @@ export const command_codeProvider: RegistryEntry = { id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 262144, maxOutputTokens: 65536, @@ -106,6 +119,7 @@ export const command_codeProvider: RegistryEntry = { id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 262144, maxOutputTokens: 65536, @@ -114,6 +128,7 @@ export const command_codeProvider: RegistryEntry = { id: "zai-org/GLM-5.1", name: "GLM-5.1 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 200000, maxOutputTokens: 32768, }, @@ -121,6 +136,7 @@ export const command_codeProvider: RegistryEntry = { id: "zai-org/GLM-5", name: "GLM-5 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 200000, maxOutputTokens: 32768, }, @@ -128,6 +144,7 @@ export const command_codeProvider: RegistryEntry = { id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 1048576, maxOutputTokens: 65536, }, @@ -135,6 +152,7 @@ export const command_codeProvider: RegistryEntry = { id: "MiniMaxAI/MiniMax-M2.5", name: "MiniMax M2.5 (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 1048576, maxOutputTokens: 65536, }, @@ -142,6 +160,7 @@ export const command_codeProvider: RegistryEntry = { id: "Qwen/Qwen3.6-Max-Preview", name: "Qwen 3.6 Max (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, contextLength: 1000000, maxOutputTokens: 32768, }, @@ -149,6 +168,7 @@ export const command_codeProvider: RegistryEntry = { id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus (CC)", supportsReasoning: true, + supportedThinkingEfforts: COMMAND_CODE_REASONING_EFFORTS, supportsVision: true, contextLength: 1000000, maxOutputTokens: 32768, diff --git a/open-sse/config/providers/registry/opencode/go/index.ts b/open-sse/config/providers/registry/opencode/go/index.ts index 54e30e57385..bf0c94ed90b 100644 --- a/open-sse/config/providers/registry/opencode/go/index.ts +++ b/open-sse/config/providers/registry/opencode/go/index.ts @@ -126,6 +126,69 @@ export const opencode_goProvider: RegistryEntry = { supportsReasoning: true, }, { id: "hy3-preview", name: "Hunyuan3 Preview" }, + // Muse Spark 1.2 Contributor — base + effort-tier aliases from the OpenCode Go + // registry (`opencode models opencode-go --verbose`; exact suffix set: + // minimal/low/medium/high/xhigh, no max). + { + id: "muse-spark-1.2-contributor", + name: "Muse Spark 1.2 Contributor", + contextLength: 1048576, + maxOutputTokens: 131072, + supportsReasoning: true, + supportsVision: true, + supportsAudio: true, + supportsVideo: true, + }, + { + id: "muse-spark-1.2-contributor-minimal", + name: "Muse Spark 1.2 Contributor (minimal effort)", + contextLength: 1048576, + maxOutputTokens: 131072, + supportsReasoning: true, + supportsVision: true, + supportsAudio: true, + supportsVideo: true, + }, + { + id: "muse-spark-1.2-contributor-low", + name: "Muse Spark 1.2 Contributor (low effort)", + contextLength: 1048576, + maxOutputTokens: 131072, + supportsReasoning: true, + supportsVision: true, + supportsAudio: true, + supportsVideo: true, + }, + { + id: "muse-spark-1.2-contributor-medium", + name: "Muse Spark 1.2 Contributor (medium effort)", + contextLength: 1048576, + maxOutputTokens: 131072, + supportsReasoning: true, + supportsVision: true, + supportsAudio: true, + supportsVideo: true, + }, + { + id: "muse-spark-1.2-contributor-high", + name: "Muse Spark 1.2 Contributor (high effort)", + contextLength: 1048576, + maxOutputTokens: 131072, + supportsReasoning: true, + supportsVision: true, + supportsAudio: true, + supportsVideo: true, + }, + { + id: "muse-spark-1.2-contributor-xhigh", + name: "Muse Spark 1.2 Contributor (xhigh effort)", + contextLength: 1048576, + maxOutputTokens: 131072, + supportsReasoning: true, + supportsVision: true, + supportsAudio: true, + supportsVideo: true, + }, // #8353: Grok 4.5 + effort tiers from the OpenCode Go registry. { id: "grok-4.5", name: "Grok 4.5", supportsReasoning: true }, { id: "grok-4.5-low", name: "Grok 4.5 (low effort)", supportsReasoning: true }, diff --git a/open-sse/config/providers/registry/opencode/index.ts b/open-sse/config/providers/registry/opencode/index.ts index 4aaa6ea046e..07f228b77e8 100644 --- a/open-sse/config/providers/registry/opencode/index.ts +++ b/open-sse/config/providers/registry/opencode/index.ts @@ -22,6 +22,26 @@ export const opencodeProvider: RegistryEntry = { supportsReasoning: true, interleavedField: "reasoning_content", }, + // #MUSE_SPARK: Muse Spark is served by OpenCode Zen ONLY on the OpenAI + // Responses API (https://opencode.ai/zen/v1/responses), not /chat/completions + // (confirmed in the official OpenCode Zen docs: https://opencode.ai/docs/zen/). + // Without targetFormat:"openai-responses" these models fall through to the + // default chat/completions pass-through and the upstream returns null/empty + // content (see issue #10867). The opencode provider is passthrough, so + // declaring them here only sets the wire format / capability flags — the + // live upstream model list already advertises both ids. + { + id: "muse-spark-1.2", + name: "Muse Spark 1.2", + supportsReasoning: true, + targetFormat: "openai-responses", + }, + { + id: "muse-spark-1.2-contributor-free", + name: "Muse Spark 1.2 Contributor Free", + supportsReasoning: true, + targetFormat: "openai-responses", + }, { id: "deepseek-v4-flash-free", name: "DeepSeek V4 Flash Free", supportsReasoning: true }, // #6998: 2026-07-14 refresh — the upstream free tier rotated its lineup; // minimax-m3-free, minimax-m2.5-free, ling-2.6-1t-free, diff --git a/open-sse/executors/base/reasoningEffort.ts b/open-sse/executors/base/reasoningEffort.ts index 9b8ffb1da03..fa416dbc1a8 100644 --- a/open-sse/executors/base/reasoningEffort.ts +++ b/open-sse/executors/base/reasoningEffort.ts @@ -275,6 +275,21 @@ export function sanitizeReasoningEffortForProvider( return stripEffortValue(b, c); } + // `minimal` is a sub-`low` reasoning tier some catalogs advertise (e.g. + // Muse Spark via models.dev) and the Codex provider accepts natively — but + // Command Code rejects it outright: + // Validation error: Invalid option: expected one of + // "low"|"medium"|"high"|"xhigh"|"max" at "params.reasoning_effort" + // Map it to the closest supported value (`low`) for command-code only; + // other providers (codex etc.) keep their native `minimal` handling. + if (provider === "command-code" && effortStr === "minimal") { + log?.info?.( + "REASONING_SANITIZE", + `${provider}/${modelStr}: mapped reasoning_effort minimal → low` + ); + return writeEffortValue(b, "low", c); + } + // Command Code accepts the literal top-tier value `max`, while the shared // standardization stage may have already represented the client's `max` as // OmniRoute's internal `xhigh`. Convert it back before the upstream request. diff --git a/open-sse/executors/commandCode.ts b/open-sse/executors/commandCode.ts index 6f7fe089405..b736e6cdf1f 100644 --- a/open-sse/executors/commandCode.ts +++ b/open-sse/executors/commandCode.ts @@ -2,7 +2,12 @@ import { randomUUID } from "node:crypto"; import { isVisionModelId } from "@/shared/constants/visionModels"; import { REGISTRY } from "../config/providerRegistry.ts"; -import { BaseExecutor, mergeUpstreamExtraHeaders, type ExecuteInput } from "./base.ts"; +import { + BaseExecutor, + mergeUpstreamExtraHeaders, + sanitizeReasoningEffortForProvider, + type ExecuteInput, +} from "./base.ts"; type JsonRecord = Record; @@ -387,6 +392,38 @@ const COMMAND_CODE_PASSTHROUGH_FIELDS = [ "extra_body", ] as const; +/** + * Command Code's /alpha/generate endpoint serves most models under a + * vendor-prefixed wire id (e.g. `xiaomi/mimo-v2.5`, `deepseek/deepseek-v4-pro`, + * `moonshotai/Kimi-K2.6`) and defaults an unprefixed id to the `anthropic:` + * provider, which 403s with "Model/provider not recognized: anthropic:". + * The command-code registry ids already carry the vendor prefix, so a bare id + * reaching the executor is an operator-set custom model (e.g. the Vision Bridge + * picker, #10809). Map the small set of documented bare ids to their + * vendor-prefixed wire form; anything with an explicit `/` (or already wired) + * passes through untouched. Kept minimal and doc-backed, mirroring the + * `CC_VISION_MODEL_PATTERNS` philosophy. + */ +const COMMAND_CODE_BARE_MODEL_VENDOR_PREFIX: Readonly> = { + // Xiaomi MiMo V2.5 — the only CC-served vision model not in the registry. + "mimo-v2.5": "xiaomi/mimo-v2.5", + "mimo-v2.5-pro": "xiaomi/mimo-v2.5-pro", +}; + +/** + * Normalize an incoming model id to the wire form Command Code's upstream + * accepts. Strips a leading provider prefix (`command-code/` / `cmd/`) that the + * pipeline may have resolved, then maps known bare ids to their + * vendor-prefixed form (see above). + */ +function normalizeCommandCodeWireModel(model: string): string { + const trimmed = String(model || "").trim(); + if (!trimmed) return trimmed; + const bare = trimmed.replace(/^(?:command-code|cmd)\//, ""); + if (bare.includes("/")) return bare; + return COMMAND_CODE_BARE_MODEL_VENDOR_PREFIX[bare] ?? bare; +} + function buildCommandCodeBody( model: string, body: unknown, @@ -398,8 +435,10 @@ function buildCommandCodeBody( // Payload rules may rewrite `body.model` (e.g. deepseek-v4-pro-max → // deepseek/deepseek-v4-pro for the command-code provider). Prefer the // rewritten value if present; fall back to the resolved combo model arg. - const resolvedModel = - typeof input.model === "string" && input.model.trim().length > 0 ? input.model : model; + // Normalize to the vendor-prefixed wire id the upstream requires (#10809). + const resolvedModel = normalizeCommandCodeWireModel( + typeof input.model === "string" && input.model.trim().length > 0 ? input.model : model + ); const converted = convertMessages(input.messages, resolvedModel, toolNameMap); const explicitSystem = typeof input.system === "string" ? input.system : ""; @@ -953,7 +992,17 @@ export class CommandCodeExecutor extends BaseExecutor { }; mergeUpstreamExtraHeaders(headers, upstreamExtraHeaders); - const { body: transformedBody, toolNameMap } = buildCommandCodeBody(model, body, stream); + // The combo/single-model dispatch boundary does not always run + // sanitizeRequestForResolvedTarget before reaching this executor (combo + // path), and Command Code rejects unsupported reasoning_effort values + // outright (e.g. "minimal" → 400 "expected one of low|medium|high|xhigh|max"). + // Sanitize here — the executor is the last line of defense for the wire body. + const sanitizedBody = sanitizeReasoningEffortForProvider(body, this.provider, model); + const { body: transformedBody, toolNameMap } = buildCommandCodeBody( + model, + sanitizedBody, + stream + ); const url = this.buildUrl(); const upstream = await fetch(url, { method: "POST", diff --git a/open-sse/executors/copilot-m365-connection.ts b/open-sse/executors/copilot-m365-connection.ts index c8251af5f4c..1463a4da632 100644 --- a/open-sse/executors/copilot-m365-connection.ts +++ b/open-sse/executors/copilot-m365-connection.ts @@ -165,8 +165,7 @@ export function resolveConnectionParams( // token is an opaque JWE with 5) is the freshest copy: the executor refreshes it // in place before resolving params, and the framework mutates it after a refresh. const credentialsJwt = - typeof credentials?.accessToken === "string" && - credentials.accessToken.split(".").length === 3 + typeof credentials?.accessToken === "string" && credentials.accessToken.split(".").length === 3 ? credentials.accessToken : ""; const accessToken = @@ -309,9 +308,7 @@ export function tokenNeedsRefresh(token: string, leadMs = M365_REFRESH_LEAD_MS): } /** The freshest readable access token for a connection (JWT column → apiKey → psd). */ -export function currentM365AccessToken( - credentials: ProviderCredentials | undefined -): string { +export function currentM365AccessToken(credentials: ProviderCredentials | undefined): string { if ( typeof credentials?.accessToken === "string" && credentials.accessToken.split(".").length === 3 @@ -393,21 +390,169 @@ export async function refreshM365AccessToken( } } -/** Flatten OpenAI messages into a single prompt (system instructions prepended). */ -export function buildPrompt(body: JsonRecord | undefined): string { - const messages = (body?.messages as Array) || []; - const systemMsgs = messages.filter((m) => m.role === "system"); - const userMsg = messages.filter((m) => m.role === "user").pop(); - const userText = - typeof userMsg?.content === "string" ? userMsg.content : JSON.stringify(userMsg?.content ?? ""); - let prompt = ""; - if (systemMsgs.length > 0) { - const sysText = systemMsgs - .map((m) => (typeof m.content === "string" ? m.content : "")) +/** A client-declared tool, normalized from the OpenAI `tools[]` entry. */ +export interface M365ToolSpec { + name: string; + description: string; + parameters: JsonRecord | null; +} + +/** + * Extract `tools` / `tool_choice` from an OpenAI chat-completion body, normalizing + * function tools into {@link M365ToolSpec}. Non-function tools and entries without + * a name are dropped (they cannot be expressed in the M365 protocol). + */ +export function extractToolSpec(body: JsonRecord | undefined): { + tools: M365ToolSpec[]; + toolChoice: unknown; +} { + const raw = Array.isArray(body?.tools) ? (body!.tools as JsonRecord[]) : []; + const tools: M365ToolSpec[] = []; + for (const t of raw) { + if (t?.type !== "function") continue; + const fn = (t.function ?? {}) as JsonRecord; + const name = typeof fn.name === "string" ? fn.name : ""; + if (!name) continue; + tools.push({ + name, + description: typeof fn.description === "string" ? fn.description : "", + parameters: + fn.parameters && typeof fn.parameters === "object" ? (fn.parameters as JsonRecord) : null, + }); + } + return { tools, toolChoice: body?.tool_choice ?? null }; +} + +/** Compact a tool result before it is folded into the flattened prompt. */ +function compactToolResult(text: string, maxChars = 4000): string { + if (text.length <= maxChars) return text; + return `${text.slice(0, maxChars)}\n…[truncated ${text.length - maxChars} chars]`; +} + +function messageText(content: unknown): string { + if (typeof content === "string") return content; + if (Array.isArray(content)) { + // Multimodal content parts: keep text parts, skip image parts (unsupported here). + return content + .map((p) => + p && typeof p === "object" && typeof (p as JsonRecord).text === "string" + ? (p as JsonRecord).text + : "" + ) .filter(Boolean) .join("\n"); - if (sysText) prompt += `[System Instructions]\n${sysText}\n\n`; } - prompt += userText; - return prompt; + return content == null ? "" : JSON.stringify(content); +} + +/** + * Flatten the FULL OpenAI message history into a single bracketed prompt — earlier + * turns, assistant replies (including `tool_calls`), and tool results, so multi-turn + * agent loops keep their context. Tool results are compacted via + * {@link compactToolResult} to keep a long loop from exhausting the turn budget. + */ +export function flattenMessages(body: JsonRecord | undefined): string { + const messages = (body?.messages as Array) || []; + const parts: string[] = []; + for (const m of messages) { + const role = typeof m.role === "string" ? m.role.toLowerCase().trim() : "user"; + const text = messageText(m.content).trim(); + if (Array.isArray(m.tool_calls) && m.tool_calls.length > 0) { + if (text) parts.push(`[${role}]\n${text}`); + parts.push(`[${role} tool_calls]\n${JSON.stringify(m.tool_calls)}`); + continue; + } + if (role === "tool") { + const id = typeof m.tool_call_id === "string" ? m.tool_call_id : ""; + parts.push(`[tool result id=${id}]\n${compactToolResult(text)}`); + continue; + } + if (!text) continue; + parts.push(`[${role}]\n${text}`); + } + return parts.join("\n\n").trim(); +} + +/** + * Wrap the flattened prompt in the community M365 tool-calling protocol: definitions + * inside a `` block, and the model answering with fenced blocks whose info + * string is the exact tool name and whose body is a JSON object of arguments. + * `tool_choice: "none"` keeps the plain prompt (no tool use requested this turn). + */ +function toolProtocolPrompt(text: string, tools: M365ToolSpec[], toolChoice: unknown): string { + if (tools.length === 0 || toolChoice === "none") { + return `Please answer the following request in full. Do not truncate or abbreviate your response.\n\n${text}`; + } + const defs = tools.map((t) => { + const params = t.parameters ? JSON.stringify(t.parameters, null, 2) : "{}"; + return `${t.name} — ${t.description}\n\`\`\`${t.name}\n${params}\n\`\`\``; + }); + return ( + `You are an execution agent operating on behalf of the application that sent this ` + + `request. The tools below are real, active, and callable right now — they were ` + + `registered by that application for this conversation. Do not analyze whether tools ` + + `are registered, available, or permitted: they are. Never state that a tool is ` + + `unavailable or that you cannot call tools.\n` + + `When the user's request requires a tool, call it by emitting one or more fenced code ` + + `blocks. Each block's info string is the exact tool name and its body is a single JSON ` + + `object of arguments. For independent operations, emit multiple blocks in one response. ` + + `Do not wrap tool calls in any other structure, and wait for the tool result before ` + + `claiming completion.\n\n\n${defs.join("\n\n")}\n\n\n${text}` + ); +} + +/** + * Flatten OpenAI messages into a single prompt (full history), and — when the + * client declared `tools` — wrap it in the M365 fenced-block tool protocol so the + * model's tool calls can be parsed back into OpenAI `tool_calls` downstream. + */ +export function buildPrompt(body: JsonRecord | undefined): string { + const { tools, toolChoice } = extractToolSpec(body); + return toolProtocolPrompt(flattenMessages(body), tools, toolChoice); +} + +/** + * Build the ROUTER-planning prompt — the strategy the substrate model actually + * complies with. Asking it to "use" a client tool gets refused ("not available in + * this chat environment") because it checks its own plugin registry; asking it to + * act as a tool-SELECTION assistant that prints a routing decision as plain text + * (`CALL_TOOL: name({...})` / `NO_TOOL_NEEDED`) bypasses that refusal entirely. + */ +export function buildRouterPrompt( + text: string, + tools: M365ToolSpec[], + toolChoice: unknown +): string { + const defs = JSON.stringify( + tools.map((t) => ({ + type: "function", + function: { name: t.name, description: t.description, parameters: t.parameters ?? {} }, + })) + ); + const choice = + typeof toolChoice === "string" && toolChoice !== "auto" && toolChoice !== "none" + ? toolChoice + : toolChoice && typeof toolChoice === "object" + ? (((toolChoice as JsonRecord).function as JsonRecord | undefined)?.name ?? "auto") + : "auto"; + let rules = + `- If a tool is needed, respond with: CALL_TOOL: tool_name({"arg1":"value1"})\n` + + `- If multiple independent tools are needed, output one CALL_TOOL line per tool\n` + + `- If no tool is needed, respond with: NO_TOOL_NEEDED\n` + + `- Only use tools from the available list above\n` + + `- Validate all arguments against the tool's schema\n` + + `- Do not invent tools that are not in the list`; + // Multi-turn: completed tool evidence in the history was already acted upon — + // re-invoking those tools would duplicate work. + if (text.includes("[tool result id=") || text.includes("[assistant tool_calls]")) { + rules += + `\n- Completed evidence must not be repeated: prior tool_calls/tool results are ` + + `already delivered, never re-invoke them\n` + + `- Only start a new tool call when fresh unfinished work remains on the current request`; + } + return ( + `You are a tool selection assistant. Based on the user request, decide which tool to call next.\n\n` + + `Available tools: ${defs}\n\nMODE: ${choice}\n\nRules:\n${rules}\n\n` + + `User request and evidence:\n${text}` + ); } diff --git a/open-sse/executors/copilot-m365-frames.ts b/open-sse/executors/copilot-m365-frames.ts index c8c6dee7be0..7260afebd7a 100644 --- a/open-sse/executors/copilot-m365-frames.ts +++ b/open-sse/executors/copilot-m365-frames.ts @@ -18,6 +18,8 @@ * accumulated — NOT incremental) → isLastUpdate:true → type:2 final → type:3 completion. */ +type JsonRecord = Record; + /** SignalR record separator (0x1e) terminating every JSON frame. */ export const RECORD_SEPARATOR = String.fromCharCode(0x1e); @@ -210,6 +212,203 @@ export interface ChatInvocationOptions { * surface omits the key entirely, so it is left out unless set (#10718). */ disconnectBehavior?: string; + /** Client-declared tool plugins (see {@link clientPlugins}); defaults to `[]`. */ + plugins?: JsonRecord[]; + /** OpenAI `tool_choice` echoed to the substrate; defaults to `null`. */ + toolChoice?: unknown; + /** Tool-use nudge sent as `customInstructions` when tools are declared. */ + customInstructions?: string; +} + +/** A client-declared tool in the normalized shape produced by `extractToolSpec`. */ +export interface M365ToolDecl { + name: string; + description: string; + parameters: JsonRecord | null; +} + +/** + * Map normalized OpenAI function tools to the M365 `plugins[]` invocation entries + * (`{Id, Source:"API", Description, Parameters}`), mirroring the community M365 + * convention. Entries without a name are skipped by the extractor upstream. + */ +export function clientPlugins(tools: M365ToolDecl[]): JsonRecord[] { + return tools.map((t) => ({ + Id: t.name, + Source: "API", + Description: t.description, + Parameters: t.parameters ?? {}, + })); +} + +/** True when `toolChoice` permits calling `name` (string / typed / "required"/"auto"). */ +function toolChoiceAllows(toolChoice: unknown, name: string): boolean { + if (toolChoice == null || toolChoice === "auto" || toolChoice === "required") return true; + if (typeof toolChoice === "string") return toolChoice === name; + const fn = (toolChoice as JsonRecord)?.function as JsonRecord | undefined; + return typeof fn?.name === "string" && fn.name === name; +} + +/** A tool call parsed from the model's fenced-block or router output. */ +export interface M365ParsedToolCall { + id: string; + type: string; + name: string; + /** JSON-stringified arguments object, as the OpenAI `tool_calls` shape expects. */ + arguments: string; +} + +const SHELL_TOOL_NAMES = ["bash", "sh", "shell", "powershell", "cmd"] as const; +const FENCED_BLOCK = /```([A-Za-z0-9_-]+)[ \t]*\r?\n([\s\S]*?)\r?\n```/g; + +/** + * Parse the model's fenced-block tool calls out of a completed turn + * (```` ```toolname\n{json args}\n``` ```` — the protocol taught by the prompt). + * Only names the client actually declared are accepted (undeclared names such as + * a hallucinated `unknown_tool` must never reach the caller), and `tool_choice` + * restrictions are enforced the same way. A shell-family block emitted for a + * DECLARED shell tool is normalized into `{command: "..."}`. + */ +export function parseFencedToolCalls( + text: string, + tools: M365ToolDecl[], + toolChoice: unknown +): M365ParsedToolCall[] { + const allowed = new Set(tools.map((t) => t.name)); + const declaredShell = SHELL_TOOL_NAMES.find((n) => allowed.has(n)); + const out: M365ParsedToolCall[] = []; + for (const m of text.matchAll(FENCED_BLOCK)) { + const name = m[1]!; + const body = m[2]!.trim(); + let parsed: unknown; + try { + parsed = JSON.parse(body); + } catch { + parsed = undefined; + } + // Shell-family blocks: keep only for a declared shell tool, normalizing a + // plain-text body (or {"command": ...}) into the canonical arguments object. + if ((SHELL_TOOL_NAMES as readonly string[]).includes(name)) { + const target = allowed.has(name) ? name : declaredShell; + if (!target) continue; + const args = + parsed && typeof parsed === "object" && "command" in (parsed as JsonRecord) + ? (parsed as JsonRecord) + : { command: body }; + out.push({ + id: `call_${crypto.randomUUID()}`, + type: "function", + name: target, + arguments: JSON.stringify(args), + }); + continue; + } + if (!allowed.has(name) || !toolChoiceAllows(toolChoice, name)) continue; + if (parsed == null || typeof parsed !== "object") continue; + out.push({ + id: `call_${crypto.randomUUID()}`, + type: "function", + name, + arguments: JSON.stringify(parsed), + }); + } + return out; +} + +/** A router-turn decision: `decided:false` means the output was unparseable. */ +export interface M365RouterDecision { + decided: boolean; + calls: M365ParsedToolCall[]; +} + +function allowedName(tools: M365ToolDecl[], name: string): boolean { + return tools.some((t) => t.name === name); +} + +function validCall( + name: string, + args: unknown, + tools: M365ToolDecl[], + toolChoice: unknown +): M365ParsedToolCall | null { + if (!name || !allowedName(tools, name) || !toolChoiceAllows(toolChoice, name)) return null; + if (!args || typeof args !== "object") return null; + return { + id: `call_${crypto.randomUUID()}`, + type: "function", + name, + arguments: JSON.stringify(args), + }; +} + +/** + * Parse the router turn's decision (`CALL_TOOL: name({...})` lines / + * `NO_TOOL_NEEDED`), validating every call against the declared tools and + * `tool_choice`. Falls back to the `{"calls":[...]}` JSON envelope. Returns + * `decided:false` when the output is neither shape, so the caller can fall + * through to a plain answer turn instead of guessing. + */ +export function parseToolRouterDecision( + text: string, + tools: M365ToolDecl[], + toolChoice: unknown +): M365RouterDecision { + const trimmed = text.trim(); + const calls: M365ParsedToolCall[] = []; + for (const line of trimmed.split(/\r?\n/)) { + const m = /^CALL_TOOL:\s*(.+)$/i.exec(line.trim()); + if (!m) continue; + const rest = m[1]!; + const start = rest.indexOf("("); + const end = rest.lastIndexOf(")"); + if (start <= 0 || end <= start) continue; + const name = rest.slice(0, start).trim(); + try { + const args = JSON.parse(rest.slice(start + 1, end)); + const call = validCall(name, args, tools, toolChoice); + if (call) calls.push(call); + } catch { + /* malformed JSON on this line — skip */ + } + } + if (calls.length > 0) return { decided: true, calls }; + if (/^no_tool_needed$/i.test(trimmed) || trimmed.toLowerCase().includes("no_tool_needed")) { + return { decided: true, calls: [] }; + } + // Fallback: the {"calls":[{"name","arguments"}]} envelope, optionally fenced. + let probe = trimmed; + const fence = probe.indexOf("```"); + if (fence >= 0) { + probe = probe + .slice(fence + 3) + .replace(/```$/, "") + .trim(); + probe = probe.replace(/^(json|JSON)\s*/, ""); + } + const start = probe.indexOf("{"); + const end = probe.lastIndexOf("}"); + if (start >= 0 && end > start) { + try { + const parsed = JSON.parse(probe.slice(start, end + 1)) as { + calls?: Array<{ name?: unknown; arguments?: unknown }>; + }; + if (Array.isArray(parsed.calls)) { + for (const c of parsed.calls) { + const call = validCall( + typeof c?.name === "string" ? c.name : "", + c?.arguments, + tools, + toolChoice + ); + if (call) calls.push(call); + } + return { decided: true, calls }; + } + } catch { + /* not JSON — undecided */ + } + } + return { decided: false, calls: [] }; } /** @@ -313,7 +512,8 @@ export function buildChatInvocation(opts: ChatInvocationOptions): Record | null): boolea return !!frame && frame.type === 3; } +/** + * Extract the error message from a `type:3` completion frame that carries one + * (`frame.error.message` / `frame.error`). A clean completion returns null — + * without this check a server-side invocation error surfaces as a silent empty + * `stop`, indistinguishable from a genuine empty reply. + */ +export function extractCompletionError(frame: Record | null): string | null { + if (!frame || frame.type !== 3) return null; + const error = frame.error; + if (!error || typeof error !== "object") return null; + const message = (error as JsonRecord).message; + return typeof message === "string" && message.length > 0 ? message : JSON.stringify(error); +} + +/** + * True for messages that carry tool/search/code PROGRESS rather than answer text + * (`messageType:"Progress"`, or the SearchResults/Code/ToolCall content types). + * Such text must never be folded into the streamed answer. + */ +function isToolProgressMessage(m: Record): boolean { + if (m.messageType === "Progress") return true; + const ct = m.contentType; + return ct === "SearchResults" || ct === "Code" || ct === "ToolCall" || ct === "EarlyProgress"; +} + +/** + * True when an update frame is a tool-progress frame — it carries Progress / + * SearchResults / Code / ToolCall messages alongside (possibly) a `writeAtCursor` + * increment that belongs to that progress, not to the answer (the browser client + * suppresses such writeAtCursor deltas; so must we). + */ +export function isToolProgressFrame(frame: Record | null): boolean { + if (!isUpdateFrame(frame)) return false; + const args = frame.arguments; + const first = Array.isArray(args) ? (args[0] as Record | undefined) : undefined; + const messages = first?.messages; + if (!Array.isArray(messages)) return false; + return messages.some( + (m) => !!m && typeof m === "object" && isToolProgressMessage(m as Record) + ); +} + /** True when an update frame is flagged as the last update of the turn. */ export function isLastUpdate(frame: Record | null): boolean { if (!isUpdateFrame(frame)) return false; @@ -366,7 +608,7 @@ export function extractBotText(frame: Record | null): string | if (!m) continue; const author = m.author; const text = m.text; - if (m.messageType === "Progress" || m.contentType === "EarlyProgress") continue; + if (isToolProgressMessage(m)) continue; if ((author === "bot" || author === undefined) && typeof text === "string" && text.length > 0) { return text; } @@ -425,6 +667,9 @@ export function accumulateBotContent( previous: string, frame: Record | null ): { delta: string; next: string } { + // A tool-progress frame's writeAtCursor belongs to the progress card (search + // queries, code interpreter output…), not to the answer text. + if (isToolProgressFrame(frame)) return { delta: "", next: previous }; const snapshot = extractBotText(frame); if (snapshot) { return { delta: incrementalDelta(previous, snapshot), next: snapshot }; diff --git a/open-sse/executors/copilot-m365-web.ts b/open-sse/executors/copilot-m365-web.ts index 5bdaf9ae9bf..4bfe4014b44 100644 --- a/open-sse/executors/copilot-m365-web.ts +++ b/open-sse/executors/copilot-m365-web.ts @@ -4,10 +4,13 @@ import { sanitizeErrorMessage } from "../utils/error.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorLog } from "./base.ts"; import { buildPrompt, + buildRouterPrompt, buildWsUrl, currentM365AccessToken, currentM365ChathubPath, decodeJwtClaims, + extractToolSpec, + flattenMessages, redactWsUrl, refreshM365AccessToken, resolveConnectionParams, @@ -16,20 +19,30 @@ import { import { accumulateBotContent, buildChatInvocation, + clientPlugins, encodeFrame, + extractCompletionError, extractFinalResultMessage, handshakeError, handshakeFrame, isCompletionFrame, isUpdateFrame, + keepaliveFrame, metricsFrame, + parseFencedToolCalls, parseFrame, + parseToolRouterDecision, resolveChatInvocationOverrides, resolveToneForModel, splitFrames, } from "./copilot-m365-frames.ts"; type JsonRecord = Record; +type M365ToolDecl = { + name: string; + description: string; + parameters: JsonRecord | null; +}; let WebSocketCtor: typeof WebSocket = WebSocket; export function __setCopilotM365WebSocketForTesting(ctor: typeof WebSocket): () => void { @@ -70,6 +83,97 @@ function errorResponse(message: string, status = 502): Response { }); } +/** Consume one wsChat SSE stream to its full text (router turns are read fully). */ +async function readSseText(stream: ReadableStream): Promise { + const reader = stream.getReader(); + const decoder = new TextDecoder(); + let fullText = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + for (const line of decoder.decode(value, { stream: true }).split("\n")) { + if (!line.startsWith("data: ")) continue; + const data = line.slice(6).trim(); + if (!data || data === "[DONE]") continue; + try { + const parsed = JSON.parse(data) as JsonRecord; + const choices = parsed.choices; + const choice = (Array.isArray(choices) ? choices[0] : undefined) as + { delta?: { content?: unknown } } | undefined; + if (typeof choice?.delta?.content === "string") fullText += choice.delta.content; + } catch { + /* skip malformed SSE lines */ + } + } + } + return fullText; +} + +/** Build the tool_calls result for a routed decision (stream + non-stream). */ +function toolCallsResult( + calls: Array<{ id: string; type: string; name: string; arguments: string }>, + opts: { stream: boolean; model: string; wsUrl: string } +) { + if (opts.stream) { + let sse = sseChunk(opts.model, { role: "assistant", content: null }); + for (let i = 0; i < calls.length; i++) { + sse += sseChunk(opts.model, { + tool_calls: [ + { + index: i, + id: calls[i]!.id, + type: calls[i]!.type, + function: { name: calls[i]!.name, arguments: calls[i]!.arguments }, + }, + ], + }); + } + sse += sseChunk(opts.model, {}, "tool_calls") + "data: [DONE]\n\n"; + return { + response: new Response(sse, { + headers: { + "Content-Type": "text/event-stream", + "Cache-Control": "no-cache", + Connection: "keep-alive", + }, + }), + url: redactWsUrl(opts.wsUrl), + headers: {}, + transformedBody: { model: opts.model, toolCalls: calls.length }, + }; + } + return { + response: new Response( + JSON.stringify({ + id: `chatcmpl-copilot-m365-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model: opts.model, + choices: [ + { + index: 0, + message: { + role: "assistant", + content: null, + tool_calls: calls.map((c) => ({ + id: c.id, + type: c.type, + function: { name: c.name, arguments: c.arguments }, + })), + }, + finish_reason: "tool_calls", + }, + ], + usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }, + }), + { headers: { "Content-Type": "application/json" } } + ), + url: redactWsUrl(opts.wsUrl), + headers: {}, + transformedBody: { model: opts.model, toolCalls: calls.length }, + }; +} + export class CopilotM365WebExecutor extends BaseExecutor { constructor() { super("copilot-m365-web", { id: "copilot-m365-web", baseUrl: "wss://substrate.office.com" }); @@ -80,12 +184,15 @@ export class CopilotM365WebExecutor extends BaseExecutor { prompt: string; model: string; tier?: string; + tools?: M365ToolDecl[]; + toolChoice?: unknown; signal?: AbortSignal; log?: ExecutorLog | null; }): Promise> { // #6210 — observability for the empty-response class. The access_token rides // in the WS query string, so every URL logged here goes through redactWsUrl(). const log = input.log ?? null; + const toolMode = (input.tools?.length ?? 0) > 0; return new ReadableStream( { start: async (controller) => { @@ -94,6 +201,11 @@ export class CopilotM365WebExecutor extends BaseExecutor { let settled = false; let buffer = ""; let previousText = ""; + // Tool-call streaming: with tools declared, content is emitted with a + // small tail holdback until a fenced block opens — from then on everything + // is buffered and resolved into `tool_calls` at finish, never as content. + let pendingTail = ""; + let fenceSeen = false; let finalResultMessage = ""; let handshakeComplete = false; @@ -112,17 +224,50 @@ export class CopilotM365WebExecutor extends BaseExecutor { if (settled) return; settled = true; cleanup(); - // Last-resort fallback (#6210): some EDU turns surface the answer only in the - // type:2 invocation result. Emit it if nothing was streamed. + // Last-resort fallback (#6210): some EDU turns surface the answer only + // in the type:2 invocation result. Treat it as the turn text. if (!previousText && finalResultMessage) { + previousText = finalResultMessage; + } + // Tool-call resolution: parse the fenced-block protocol out of the + // completed turn and, when the model called declared tools, close the + // stream with OpenAI `tool_calls` instead of plain content. + const calls = toolMode + ? parseFencedToolCalls(previousText, input.tools ?? [], input.toolChoice) + : []; + if (calls.length > 0) { controller.enqueue( - encoder.encode(sseChunk(input.model, { content: finalResultMessage })) + encoder.encode(sseChunk(input.model, { role: "assistant", content: null })) ); - } else if (!previousText && !finalResultMessage) { + for (let i = 0; i < calls.length; i++) { + const call = calls[i]!; + controller.enqueue( + encoder.encode( + sseChunk(input.model, { + tool_calls: [ + { + index: i, + id: call.id, + type: call.type, + function: { name: call.name, arguments: call.arguments }, + }, + ], + }) + ) + ); + } + controller.enqueue(encoder.encode(sseChunk(input.model, {}, "tool_calls"))); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + return; + } + if (!previousText) { // #7858 — a turn that completed with no content in ANY known shape is // indistinguishable, from the outside, from a genuine successful-but-empty // reply. Fail loudly instead of a silent `stop`, per Hard Rule #12. - const tierNote = input.tier ? `resolved tier: ${input.tier}` : "resolved tier: individual (default)"; + const tierNote = input.tier + ? `resolved tier: ${input.tier}` + : "resolved tier: individual (default)"; const message = sanitizeErrorMessage( `Microsoft 365 Copilot turn completed with no content in any known frame ` + `shape (${tierNote}). Possible causes: an unrecognized frame shape for ` + @@ -134,6 +279,11 @@ export class CopilotM365WebExecutor extends BaseExecutor { controller.close(); return; } + // No tool calls: flush any holdback tail as ordinary content and stop. + if (pendingTail) { + controller.enqueue(encoder.encode(sseChunk(input.model, { content: pendingTail }))); + pendingTail = ""; + } controller.enqueue(encoder.encode(sseChunk(input.model, {}, "stop"))); controller.enqueue(encoder.encode("data: [DONE]\n\n")); controller.close(); @@ -144,7 +294,9 @@ export class CopilotM365WebExecutor extends BaseExecutor { settled = true; cleanup(); const message = sanitizeErrorMessage(reason); - controller.enqueue(encoder.encode(`data: ${JSON.stringify({ error: { message } })}\n\n`)); + controller.enqueue( + encoder.encode(`data: ${JSON.stringify({ error: { message } })}\n\n`) + ); controller.close(); }; @@ -194,6 +346,18 @@ export class CopilotM365WebExecutor extends BaseExecutor { isStartOfSession: true, ...overrides, tone, + // Declare the client's tools natively too (plugins + toolChoice + + // a customInstructions nudge); the fenced-block protocol in the + // prompt remains the parseable path. + ...(toolMode + ? { + plugins: clientPlugins(input.tools ?? []), + toolChoice: input.toolChoice ?? null, + customInstructions: + "You have access to real tools provided by the calling application. " + + "Call tools directly when needed. Do not say tools are unavailable.", + } + : {}), }) ); // #10718 — the invocation and its type:1 Metrics follow-up must land @@ -233,6 +397,17 @@ export class CopilotM365WebExecutor extends BaseExecutor { continue; } + // SignalR keepalive: the server pings with type:6 and expects the + // exact echo back, or it drops the socket mid-turn on long agentic runs. + if (frame?.type === 6) { + try { + ws?.send(keepaliveFrame()); + } catch { + /* socket already closing — the close handler finishes the stream */ + } + continue; + } + const { delta, next } = accumulateBotContent(previousText, frame); if (!delta && next === previousText) { // #7858 AC2/AC3 — log unrecognized-shape update frames by KEY only, so @@ -243,7 +418,25 @@ export class CopilotM365WebExecutor extends BaseExecutor { } previousText = next; if (delta) { - controller.enqueue(encoder.encode(sseChunk(input.model, { content: delta }))); + if (!toolMode) { + controller.enqueue(encoder.encode(sseChunk(input.model, { content: delta }))); + } else if (!fenceSeen) { + // Hold back a 12-char tail so a "```" opener straddling a chunk + // boundary is never emitted as content; once any fence opens, + // buffer everything for the finish-time tool-call resolution. + pendingTail += delta; + if (pendingTail.includes("```")) { + fenceSeen = true; + } else if (pendingTail.length > 12) { + const cut = pendingTail.length - 12; + controller.enqueue( + encoder.encode( + sseChunk(input.model, { content: pendingTail.slice(0, cut) }) + ) + ); + pendingTail = pendingTail.slice(cut); + } + } } const finalMsg = extractFinalResultMessage(frame); @@ -251,6 +444,16 @@ export class CopilotM365WebExecutor extends BaseExecutor { finalResultMessage = finalMsg; } + // A type:3 carrying an error is a FAILED turn; without this it + // would finish() into a silent empty stop. + const completionError = extractCompletionError(frame); + if (completionError) { + clearTimeout(timeout); + log?.debug?.("M365_WS", `completion error: ${completionError}`); + abort(`Microsoft 365 Copilot invocation failed: ${completionError}`); + return; + } + if (isCompletionFrame(frame)) { clearTimeout(timeout); finish(); @@ -310,8 +513,7 @@ export class CopilotM365WebExecutor extends BaseExecutor { const current = currentM365AccessToken(credentials); if (current && !tokenNeedsRefresh(current)) return; - const tid = - decodeJwtClaims(current)?.tid || (typeof psd.tid === "string" ? psd.tid : "") || ""; + const tid = decodeJwtClaims(current)?.tid || (typeof psd.tid === "string" ? psd.tid : "") || ""; const result = await refreshM365AccessToken(refreshToken, tid, log ?? undefined); if ("error" in result) { // Fall through with the existing token — the WS layer will surface the failure. @@ -355,7 +557,19 @@ export class CopilotM365WebExecutor extends BaseExecutor { const body = input.body as JsonRecord | undefined; const model = input.model || (body?.model as string) || "copilot-m365"; const stream = input.stream !== false; - const prompt = buildPrompt(body).trim(); + const { tools, toolChoice } = extractToolSpec(body); + const routerActive = tools.length > 0 && toolChoice !== "none"; + // Router planning: the router turn decides tool use; the answer turn (when the + // router selects none) must be RE-FRAMED as an answer request — a raw history + // continuation makes the model keep emitting the router's decision format. + const flat = flattenMessages(body); + const prompt = ( + routerActive + ? "Please answer the following request in full, using the tool results already " + + "provided in the conversation. Do not output tool-routing decisions.\n\n" + + flat + : buildPrompt(body) + ).trim(); if (!prompt) { return { @@ -383,13 +597,44 @@ export class CopilotM365WebExecutor extends BaseExecutor { } const wsUrl = buildWsUrl(connectionParams); + let answerWsUrl: string | null = null; try { + // Router planning turn — ask the model as a tool-SELECTION assistant. Asking + // it to "use" a client tool gets refused (it checks its own plugin registry); + // printing a routing decision as text bypasses that refusal. + if (routerActive) { + const routerStream = await this.wsChat({ + wsUrl, + prompt: buildRouterPrompt(flat, tools, toolChoice), + model, + tier: connectionParams.tier, + signal: input.signal ?? undefined, + log: input.log, + }); + const routerText = await readSseText(routerStream); + const decision = parseToolRouterDecision(routerText, tools, toolChoice); + input.log?.debug?.( + "M365_TOOLS", + `router decided=${decision.decided} calls=${decision.calls.length}` + ); + if (decision.decided && decision.calls.length > 0) { + return toolCallsResult(decision.calls, { stream, model, wsUrl }); + } + // No tool needed (or unparseable): answer in a FRESH conversation below. + // Reusing the router's ConversationId makes the answer turn a continuation + // of the routing dialog, and the model keeps emitting the router's decision + // format (NO_TOOL_NEEDED) as the answer. + answerWsUrl = buildWsUrl(connectionParams); + } + const wsStream = await this.wsChat({ - wsUrl, + wsUrl: answerWsUrl ?? wsUrl, prompt, model, tier: connectionParams.tier, + tools, + toolChoice, signal: input.signal ?? undefined, log: input.log, }); @@ -412,6 +657,12 @@ export class CopilotM365WebExecutor extends BaseExecutor { const reader = wsStream.getReader(); const decoder = new TextDecoder(); let fullText = ""; + const toolCalls: Array<{ + id: string; + type: string; + name: string; + arguments: string; + }> = []; while (true) { const { done, value } = await reader.read(); if (done) break; @@ -421,14 +672,60 @@ export class CopilotM365WebExecutor extends BaseExecutor { if (!data || data === "[DONE]") continue; try { const parsed = JSON.parse(data); - const content = parsed.choices?.[0]?.delta?.content; + const choice = parsed.choices?.[0]; + const content = choice?.delta?.content; if (typeof content === "string") fullText += content; + for (const tc of choice?.delta?.tool_calls ?? []) { + toolCalls.push({ + id: String(tc.id ?? ""), + type: String(tc.type ?? "function"), + name: String(tc.function?.name ?? ""), + arguments: String(tc.function?.arguments ?? "{}"), + }); + } } catch { /* skip malformed SSE lines */ } } } + // Tool-call turn: content stops at the first fence, the calls ride in + // `tool_calls` with finish_reason "tool_calls" (OpenAI agentic-loop shape). + if (toolCalls.length > 0) { + const fenceIndex = fullText.indexOf("```"); + const content = fenceIndex > 0 ? fullText.slice(0, fenceIndex).trim() : null; + return { + response: new Response( + JSON.stringify({ + id: `chatcmpl-copilot-m365-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model, + choices: [ + { + index: 0, + message: { + role: "assistant", + content, + tool_calls: toolCalls.map((c) => ({ + id: c.id, + type: c.type, + function: { name: c.name, arguments: c.arguments }, + })), + }, + finish_reason: "tool_calls", + }, + ], + usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }, + }), + { headers: { "Content-Type": "application/json" } } + ), + url: redactWsUrl(answerWsUrl ?? wsUrl), + headers: {}, + transformedBody: { model, toolCalls: toolCalls.length }, + }; + } + return { response: new Response( JSON.stringify({ diff --git a/open-sse/executors/cursor.ts b/open-sse/executors/cursor.ts index bc9ade9d27a..ebcfe053bd2 100644 --- a/open-sse/executors/cursor.ts +++ b/open-sse/executors/cursor.ts @@ -82,6 +82,7 @@ import { visibleComposerContentFromThinking, composerReasoningRemainder, } from "./cursor/composer.ts"; +import { CursorServerConfigError, resolveCursorAgentUrl } from "./cursor/agentEndpoint.ts"; import { getActiveSyncedCatalog } from "../../src/lib/db/models/activeSyncedCatalog.ts"; // Composer helpers re-exported for external importers (tests). export { @@ -193,10 +194,6 @@ function buildExecRejection(event: ExecServerEvent): Buffer | null { } } -const CURSOR_AGENT_HOST = "agentn.global.api5.cursor.sh"; -const CURSOR_AGENT_PATH = "/agent.v1.AgentService/Run"; -const CURSOR_AGENT_URL = `https://${CURSOR_AGENT_HOST}${CURSOR_AGENT_PATH}`; - // Detect cloud environment (Edge runtime, Cloudflare Workers, etc.) const isCloudEnv = () => { if (typeof caches !== "undefined" && typeof caches === "object") return true; @@ -718,7 +715,7 @@ export class CursorExecutor extends BaseExecutor { } buildUrl() { - return CURSOR_AGENT_URL; + return PROVIDERS.cursor.baseUrl; } /** @@ -1211,10 +1208,40 @@ export class CursorExecutor extends BaseExecutor { } async execute({ model, body, stream, credentials, signal, log, upstreamExtraHeaders }) { - const url = this.buildUrl(); + const fallbackUrl = this.buildUrl(); const executionCredentials = await this.resolveExecutionCredentials(credentials); if (executionCredentials instanceof Response) { - return { response: executionCredentials, url, headers: {}, transformedBody: body }; + return { + response: executionCredentials, + url: fallbackUrl, + headers: {}, + transformedBody: body, + }; + } + let url: string; + try { + url = await resolveCursorAgentUrl(executionCredentials, signal); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + const headers = this.buildHeaders(executionCredentials); + return { + response: new Response( + JSON.stringify({ + error: { + message: sanitizeErrorMessage(message), + type: "connection_error", + code: "", + }, + }), + { + status: err instanceof CursorServerConfigError ? err.status : HTTP_STATUS.SERVER_ERROR, + headers: { "Content-Type": "application/json" }, + } + ), + url: fallbackUrl, + headers, + transformedBody: body, + }; } const headers = this.buildHeaders(executionCredentials); mergeUpstreamExtraHeaders(headers, upstreamExtraHeaders); diff --git a/open-sse/executors/cursor/agentEndpoint.ts b/open-sse/executors/cursor/agentEndpoint.ts new file mode 100644 index 00000000000..e1168df5b76 --- /dev/null +++ b/open-sse/executors/cursor/agentEndpoint.ts @@ -0,0 +1,121 @@ +import { createHmac } from "node:crypto"; + +import { mergeAbortSignals, type ProviderCredentials } from "../base.ts"; +import { stripCursorOAuthTokenPrefix } from "../../services/cursorApiKeyAuth.ts"; +import { + formatCursorAgentClientVersion, + getCursorAgentCliVersion, +} from "../../utils/cursorAgentCliVersion.ts"; +import { decodeFields } from "../../utils/cursorAgentProtobuf/wire.ts"; + +const CURSOR_API_URL = "https://api2.cursor.sh"; +const CURSOR_SERVER_CONFIG_PATH = "/aiserver.v1.ServerConfigService/GetServerConfig"; +const CURSOR_AGENT_PATH = "/agent.v1.AgentService/Run"; +const CURSOR_SERVER_CONFIG_TIMEOUT_MS = 10_000; +const CURSOR_AGENT_URL_CACHE_TTL_MS = 60 * 60 * 1000; +const CURSOR_AGENT_URL_CACHE_LIMIT = 1_000; + +type CursorAgentUrls = { agentUrl: string; agentnUrl: string }; +type CursorAgentUrlCacheEntry = CursorAgentUrls & { expiresAt: number }; +const cursorAgentUrlCache = new Map(); + +/** Reports an HTTP error from Cursor server-config discovery. */ +export class CursorServerConfigError extends Error { + constructor( + message: string, + readonly status: number + ) { + super(message); + } +} + +function validateCursorAgentUrl(value: string): string { + const url = new URL(value); + const isCursorAgentHost = + url.hostname === "api5.cursor.sh" || url.hostname.endsWith(".api5.cursor.sh"); + if ( + url.protocol !== "https:" || + !isCursorAgentHost || + url.username || + url.password || + url.search || + url.hash + ) { + throw new Error("Cursor server config included an invalid Agent URL"); + } + return url.origin; +} + +function parseCursorAgentUrls(payload: Buffer): CursorAgentUrls { + const agentUrlConfig = decodeFields(payload).find( + (field) => field.fieldNumber === 27 && field.wireType === 2 + ); + if (!agentUrlConfig || agentUrlConfig.wireType !== 2) { + throw new Error("Cursor server config did not include Agent URLs"); + } + const fields = decodeFields(agentUrlConfig.bytes); + const agentUrl = fields.find((field) => field.fieldNumber === 1 && field.wireType === 2); + const agentnUrl = fields.find((field) => field.fieldNumber === 2 && field.wireType === 2); + if (!agentUrl || agentUrl.wireType !== 2 || !agentnUrl || agentnUrl.wireType !== 2) { + throw new Error("Cursor server config included incomplete Agent URLs"); + } + return { + agentUrl: validateCursorAgentUrl(agentUrl.bytes.toString("utf8")), + agentnUrl: validateCursorAgentUrl(agentnUrl.bytes.toString("utf8")), + }; +} + +async function fetchCursorAgentUrls( + accessToken: string, + signal?: AbortSignal | null +): Promise { + const timeoutSignal = AbortSignal.timeout(CURSOR_SERVER_CONFIG_TIMEOUT_MS); + const response = await fetch(`${CURSOR_API_URL}${CURSOR_SERVER_CONFIG_PATH}`, { + method: "POST", + headers: { + authorization: `Bearer ${accessToken}`, + "connect-protocol-version": "1", + "content-type": "application/proto", + "user-agent": "connect-es/1.6.1", + "x-cursor-client-type": "cli", + "x-cursor-client-version": formatCursorAgentClientVersion(getCursorAgentCliVersion()), + }, + body: Buffer.alloc(0), + signal: signal ? mergeAbortSignals(signal, timeoutSignal) : timeoutSignal, + }); + if (!response.ok) { + throw new CursorServerConfigError( + `Cursor server config request failed with status ${response.status}`, + response.status + ); + } + return parseCursorAgentUrls(Buffer.from(await response.arrayBuffer())); +} + +/** Resolve the Agent RPC URL that Cursor assigned to this connection. */ +export async function resolveCursorAgentUrl( + credentials: ProviderCredentials, + signal?: AbortSignal | null +): Promise { + const accessToken = stripCursorOAuthTokenPrefix(credentials.accessToken || ""); + if (!accessToken) throw new Error("Cursor access token is required"); + const cacheKey = + `${credentials.connectionId || "anonymous"}:` + + createHmac("sha256", "omniroute-cursor-agent-url-cache-v1").update(accessToken).digest("hex"); + const now = Date.now(); + let urls = cursorAgentUrlCache.get(cacheKey); + if (!urls || urls.expiresAt <= now) { + const fetched = await fetchCursorAgentUrls(accessToken, signal); + urls = { ...fetched, expiresAt: now + CURSOR_AGENT_URL_CACHE_TTL_MS }; + if ( + !cursorAgentUrlCache.has(cacheKey) && + cursorAgentUrlCache.size >= CURSOR_AGENT_URL_CACHE_LIMIT + ) { + const oldestKey = cursorAgentUrlCache.keys().next().value as string | undefined; + if (oldestKey !== undefined) cursorAgentUrlCache.delete(oldestKey); + } + cursorAgentUrlCache.set(cacheKey, urls); + } + const ghostMode = credentials.providerSpecificData?.ghostMode !== false; + return `${ghostMode ? urls.agentUrl : urls.agentnUrl}${CURSOR_AGENT_PATH}`; +} diff --git a/open-sse/executors/forceResponsesUpstream.ts b/open-sse/executors/forceResponsesUpstream.ts index 0c545d980e6..4de8961a3b5 100644 --- a/open-sse/executors/forceResponsesUpstream.ts +++ b/open-sse/executors/forceResponsesUpstream.ts @@ -28,6 +28,17 @@ export function shouldForceResponsesUpstream( const providerSpecificData = credentials?.providerSpecificData ?? null; if (providerSpecificData?._omnirouteForceResponsesUpstream === true) return true; if (getOpenAICompatibleType(provider, providerSpecificData) === "responses") return false; + // apiType="chat" means the operator explicitly chose the chat/completions + // wire. Don't second-guess that choice by forcing /responses just because the + // body carries namespace tools — the standard namespace→flatten path + // (openai-responses.ts) handles those correctly for chat backends. + if ( + providerSpecificData && + typeof providerSpecificData.apiType === "string" && + providerSpecificData.apiType === "chat" + ) { + return false; + } const hasResponsesShape = body.input !== undefined || diff --git a/open-sse/executors/freebuff.ts b/open-sse/executors/freebuff.ts index bc2de30f632..c2cf5bdbc3c 100644 --- a/open-sse/executors/freebuff.ts +++ b/open-sse/executors/freebuff.ts @@ -1,3 +1,5 @@ +import { randomInt } from "node:crypto"; + import { BaseExecutor, type ExecuteInput, @@ -20,7 +22,7 @@ function generateClientSessionId(): string { const alphabet = "0123456789abcdefghijklmnopqrstuvwxyz"; let out = ""; for (let i = 0; i < 13; i++) { - out += alphabet[Math.floor(Math.random() * alphabet.length)]; + out += alphabet[randomInt(alphabet.length)]; } return out; } diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 98e2112c707..6318aaab2e5 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -1,4 +1,5 @@ import { randomUUID } from "node:crypto"; +import type { KeyHealth } from "../services/apiKeyRotator.ts"; import { DefaultExecutor } from "./default.ts"; import { @@ -268,8 +269,10 @@ export class GlmExecutor extends DefaultExecutor { stream = true, _clientHeaders?: Record | null, _model?: string, - transport: GlmTransport = getGlmTransport(credentials.providerSpecificData) + _health?: unknown, + _body?: unknown ): Record { + const transport: GlmTransport = getGlmTransport(credentials.providerSpecificData); if (transport === "openai") { return buildGlmCodingHeaders(getEffectiveKey(credentials), stream); } @@ -396,13 +399,7 @@ export class GlmExecutor extends DefaultExecutor { ): Promise { const credentials = input.credentials; const url = buildGlmChatUrl(credentials?.providerSpecificData, transport, this.config.baseUrl); - const headers = this.buildHeaders( - credentials, - input.stream, - input.clientHeaders, - input.model, - transport - ); + const headers = this.buildHeaders(credentials, input.stream, input.clientHeaders, input.model); applyConfiguredUserAgent(headers, credentials.providerSpecificData); mergeUpstreamExtraHeaders(headers, input.upstreamExtraHeaders); diff --git a/open-sse/executors/kimi-web.ts b/open-sse/executors/kimi-web.ts index aaf2b32c69b..8f9c13c3328 100644 --- a/open-sse/executors/kimi-web.ts +++ b/open-sse/executors/kimi-web.ts @@ -29,6 +29,7 @@ import { sanitizeErrorMessage, } from "../utils/error.ts"; import { extractKimiAccessToken } from "@/lib/providers/webCookieAuth"; +import { exchangeKimiRefreshToken } from "@/lib/kimi/tokenRefresh"; import { type KimiWebModelConfig, resolveKimiWebContextLength, @@ -38,7 +39,23 @@ import { export { extractKimiAccessToken }; -const BASE_URL = "https://www.kimi.com"; +export function getKimiWebBaseUrl(): string { + const envUrl = process.env.KIMI_WEB_BASE_URL?.trim(); + if (envUrl) { + return envUrl.replace(/\/+$/, ""); + } + return "https://www.kimi.ai"; +} + +export function getKimiWebChatUrl(): string { + const envChat = process.env.KIMI_WEB_CHAT_URL?.trim(); + if (envChat) { + return envChat; + } + return `${getKimiWebBaseUrl()}/apiv2/kimi.gateway.chat.v1.ChatService/Chat`; +} + +const BASE_URL = "https://www.kimi.ai"; const CHAT_URL = `${BASE_URL}/apiv2/kimi.gateway.chat.v1.ChatService/Chat`; const USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36"; @@ -305,7 +322,7 @@ export class KimiWebExecutor extends BaseExecutor { const bodyObj = (body || {}) as Record; const rawCredential = String(credentials?.accessToken || credentials?.apiKey || "").trim(); - const accessToken = extractKimiAccessToken(rawCredential); + let accessToken = extractKimiAccessToken(rawCredential); if (!accessToken) { return makeErrorResult( 400, @@ -389,6 +406,29 @@ export class KimiWebExecutor extends BaseExecutor { ); } + if (upstream.status === 401) { + const refreshToken = + credentials?.refreshToken || credentials?.providerSpecificData?.refreshToken; + if (refreshToken && typeof refreshToken === "string") { + const refreshRes = await exchangeKimiRefreshToken( + refreshToken, + getKimiWebBaseUrl() + ); + if (refreshRes.success && refreshRes.accessToken) { + accessToken = refreshRes.accessToken; + const retryHeaders = this.buildKimiHeaders(accessToken); + try { + upstream = await fetch(CHAT_URL, { + method: "POST", + headers: retryHeaders, + body: new Uint8Array(framedBody), + signal, + }); + } catch {} + } + } + } + if (!upstream.ok) { const errText = await upstream.text().catch(() => ""); return makeErrorResult( diff --git a/open-sse/executors/muse-spark-web.ts b/open-sse/executors/muse-spark-web.ts index f93191c3124..a8052c4c657 100644 --- a/open-sse/executors/muse-spark-web.ts +++ b/open-sse/executors/muse-spark-web.ts @@ -1070,7 +1070,7 @@ async function wsChat( const fail = (error: string) => finish({ content: "", deltas: [], error }); - timeout = setTimeout(() => fail("Meta AI WebSocket timed out"), 30000); + timeout = setTimeout(() => fail(`Meta AI WS timed out (readyState=${ws.readyState})`), 30000); abortHandler = () => fail("Request aborted"); signal?.addEventListener("abort", abortHandler, { once: true }); diff --git a/open-sse/executors/opencode.ts b/open-sse/executors/opencode.ts index 6f7a08fbaf4..51419ead9c5 100644 --- a/open-sse/executors/opencode.ts +++ b/open-sse/executors/opencode.ts @@ -71,7 +71,8 @@ const OPENCODE_FREE_MODELS = new Set([ * `opencode models opencode-go --verbose`; MiniMax M3 excluded — different * thinking-mode mapping): * grok-4.5 low/medium/high; hy3 none/low/high; kimi-k3 max; - * qwen3.6-plus / qwen3.7-max / qwen3.7-plus high/max + * qwen3.6-plus / qwen3.7-max / qwen3.7-plus high/max; + * muse-spark-1.2-contributor minimal/low/medium/high/xhigh (no max) */ const EFFORT_TIERS: Record = { "deepseek-v4-pro": EFFORT_LEVELS, @@ -84,6 +85,7 @@ const EFFORT_TIERS: Record = { "qwen3.6-plus": ["high", "max"], "qwen3.7-max": ["high", "max"], "qwen3.7-plus": ["high", "max"], + "muse-spark-1.2-contributor": ["minimal", "low", "medium", "high", "xhigh"], }; /** diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index beb842bf2d1..e2a3d12a6b6 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -347,6 +347,7 @@ import { computeBillableTokens, normalizeExecutorResult, executeWithUpstreamStartTimeout, + resolveConnectionTimeoutMs, } from "./chatCore/upstreamTimeouts.ts"; import { getModelNormalizeToolCallId, getModelPreserveOpenAIDeveloperRole } from "@/lib/db/models"; import { getProviderCredentials, extractSessionAffinityKey } from "@/sse/services/auth"; @@ -2640,12 +2641,16 @@ export async function handleChatCore({ // no-op. #7694: `modelInfo.resolvedThinkingEffort` — set when the request's model // id carried a `/-{effort}` synced-model alias suffix // (`src/sse/services/model.ts`) — takes priority over the static per-model default. - // See open-sse/services/defaultReasoningEffort.ts. + // The synced catalog's vendor-declared `defaultThinkingEffort` (OpenRouter + // `reasoning.default_effort`, captured by `detectDefaultThinkingEffort`) is the + // lowest-priority default: it only fires when neither the suffix alias nor a + // static operator default exists. See open-sse/services/defaultReasoningEffort.ts. if (targetFormat === FORMATS.OPENAI) { translatedBody = applyDefaultReasoningEffort( translatedBody, finalModelToUpstream, - (modelInfo as { resolvedThinkingEffort?: string })?.resolvedThinkingEffort + (modelInfo as { resolvedThinkingEffort?: string })?.resolvedThinkingEffort, + (modelInfo as { defaultThinkingEffort?: string })?.defaultThinkingEffort ); } } @@ -3065,6 +3070,9 @@ export async function handleChatCore({ executor, provider, model: modelToCall, + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), signal: streamController.signal, log, execute: (signal) => @@ -3368,6 +3376,9 @@ export async function handleChatCore({ executor, provider, model: modelToCall, + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), signal: streamController.signal, log, execute: (signal) => diff --git a/open-sse/handlers/chatCore/clientUsageBuffer.ts b/open-sse/handlers/chatCore/clientUsageBuffer.ts index 4c7b1e48dcf..b34abba1f60 100644 --- a/open-sse/handlers/chatCore/clientUsageBuffer.ts +++ b/open-sse/handlers/chatCore/clientUsageBuffer.ts @@ -23,6 +23,7 @@ import { filterUsageForFormat as defaultFilterUsage, estimateUsage as defaultEstimateUsage, sanitizeProviderUsageForRequest, + type UsageLike, } from "../../utils/usageTracking.ts"; type ResponseLike = @@ -106,15 +107,18 @@ export function applyClientUsageBuffer( const { preserveContextBudgetInVisibleUsage = false } = options; if (translatedResponse?.usage) { translatedResponse.usage = sanitizeProviderUsageForRequest( - translatedResponse.usage, + translatedResponse.usage as UsageLike, body, clientResponseFormat ); } // Add buffer and filter usage for client (to prevent CLI context errors) - if (translatedResponse?.usage && !isEmptyUsage(translatedResponse.usage)) { - const buffered = deps.addBufferToUsage(translatedResponse.usage) as Record; + if (translatedResponse?.usage && !isEmptyUsage(translatedResponse.usage as UsageLike)) { + const buffered = deps.addBufferToUsage(translatedResponse.usage as UsageLike) as Record< + string, + unknown + >; if (preserveContextBudgetInVisibleUsage) { foldContextBudgetIntoVisibleUsage(buffered); } diff --git a/open-sse/handlers/chatCore/upstreamTimeouts.ts b/open-sse/handlers/chatCore/upstreamTimeouts.ts index 9f0ace2b0af..b1a8da548d0 100644 --- a/open-sse/handlers/chatCore/upstreamTimeouts.ts +++ b/open-sse/handlers/chatCore/upstreamTimeouts.ts @@ -9,6 +9,7 @@ import { getLoggedOutputTokens, getReasoningTokens, } from "@/lib/usage/tokenAccounting"; +import { MAX_PROVIDER_SPECIFIC_TIMEOUT_MS } from "@/shared/validation/providerSpecificData"; export function createBodyTimeoutError(timeoutMs: number): Error { const err = new Error(`Response body read timeout after ${timeoutMs}ms`); @@ -89,14 +90,44 @@ function resolveProviderTimeoutMs(executor: unknown): number { } } +/** Per-connection operator timeout tier: reads + * `providerSpecificData.timeoutMs`, bounded to 1..86_400_000 ms. + * Returns undefined when absent or invalid so the chain falls through. */ +export function resolveConnectionTimeoutMs(psd: unknown): number | undefined { + const timeoutMs = (psd as Record | null | undefined)?.timeoutMs; + if (typeof timeoutMs !== "number" || !Number.isFinite(timeoutMs)) return undefined; + const floored = Math.floor(timeoutMs); + if (floored < 1 || floored > MAX_PROVIDER_SPECIFIC_TIMEOUT_MS) return undefined; + return floored; +} + /** * Resolves the upstream header-response timeout in precedence order: + * connection-level override (`providerSpecificData.timeoutMs`) → * model-level override (registry `RegistryModel.timeoutMs`) → provider-level * override (`executor.getTimeoutMs()`) → global `FETCH_TIMEOUT_MS` default. * `provider`/`model` are optional so existing single-argument call sites * keep resolving to the provider/global chain unchanged (#6354). */ -export function getExecutorTimeoutMs(executor: unknown, provider?: string, model?: string): number { +export function getExecutorTimeoutMs( + executor: unknown, + provider?: string, + model?: string, + connectionTimeoutMs?: number +): number { + if ( + typeof connectionTimeoutMs === "number" && + Number.isFinite(connectionTimeoutMs) && + connectionTimeoutMs > 0 + ) { + // Defensive backstop for direct callers: resolveConnectionTimeoutMs is the + // gate (it rejects out-of-range values so the chain falls through); this + // clamp only caps values a future caller could pass unvetted. + return Math.min( + Math.max(0, Math.floor(connectionTimeoutMs)), + MAX_PROVIDER_SPECIFIC_TIMEOUT_MS + ); + } const modelOverride = resolveModelTimeoutOverride(provider, model); if (modelOverride !== undefined) return modelOverride; return resolveProviderTimeoutMs(executor); @@ -196,6 +227,7 @@ export async function executeWithUpstreamStartTimeout({ executor, provider, model, + connectionTimeoutMs, signal, log, execute, @@ -203,11 +235,12 @@ export async function executeWithUpstreamStartTimeout({ executor: unknown; provider: string; model: string; + connectionTimeoutMs?: number; signal: AbortSignal; log?: { warn?: (tag: string, message: string) => void } | null; execute: (signal: AbortSignal) => Promise; }): Promise { - const timeoutMs = getExecutorTimeoutMs(executor, provider, model); + const timeoutMs = getExecutorTimeoutMs(executor, provider, model, connectionTimeoutMs); if (timeoutMs <= 0) return execute(signal); if (signal.aborted) throw createAbortError(signal); diff --git a/open-sse/services/autoCombo/chaosEngine.ts b/open-sse/services/autoCombo/chaosEngine.ts index 89813fe48f8..32f5b09f482 100644 --- a/open-sse/services/autoCombo/chaosEngine.ts +++ b/open-sse/services/autoCombo/chaosEngine.ts @@ -180,7 +180,22 @@ function dispatchOnePanelModel(opts: { log?.info?.( `CHAOS panel ${index} (${model}) ok=${res.ok} status=${res.status} textLen=${text.length}` ); - const part: ChaosPart = { model, index, ok: true, text }; + // G5b: honor the upstream response status — a 4xx/5xx is a panel FAILURE, + // not a success (previously ok:true was hardcoded, so an all-error panel + // never reached the all-failed branch and the error text was streamed as + // if it were a successful answer). + if (res.ok) { + const part: ChaosPart = { model, index, ok: true, text }; + await onResult?.(part); + return part; + } + const part: ChaosPart = { + model, + index, + ok: false, + text: "", + error: `upstream ${res.status}: ${text.slice(0, 200) || res.statusText || "error"}`, + }; await onResult?.(part); return part; } catch (err) { @@ -466,8 +481,17 @@ export async function handleChaosChat(opts: { } if (successes.length === 0) { - const errText = "All chaos panel models failed"; - await safeEnqueue(chatChunk(chunkId, panelToDispatch[0] ?? "", errText)); + // G5 (silent-stop fix): make an all-panel failure visible server-side. + // The status stays 200 (SSE envelope must stay well-formed), but the + // failure is now logged with the per-model errors so operators can see + // why the chaos panel produced nothing. + const modelErrors = allParts.map((p) => `${p.model}: ${p.error ?? "unknown"}`).join(" | "); + log?.warn?.( + "CHAOS", + `All chaos panel models failed for ${comboName ?? "panel"}: ${modelErrors}` + ); + const errText = `All chaos panel models failed — ${modelErrors}`; + await safeEnqueue(chatChunk(chunkId, panelToDispatch[0] ?? panel[0] ?? "", errText)); await safeEnqueue(SSE_DONE); await enqueueChain; closed = true; diff --git a/open-sse/services/autoCombo/pipelineRouter.ts b/open-sse/services/autoCombo/pipelineRouter.ts index 5fbc9eea025..bc2ce3c5d5d 100644 --- a/open-sse/services/autoCombo/pipelineRouter.ts +++ b/open-sse/services/autoCombo/pipelineRouter.ts @@ -343,6 +343,17 @@ export async function handlePipelineCombo({ } } + // G6 (silent-stop fix): if the reflection loop burned its retry budget and the + // verdict is still "fail", the fall-through below returns a FAILED result + // indistinguishable from a first-attempt failure. Surface it loudly so the + // caller (and operator logs) can tell "retries exhausted" apart. + if (result.reflectVerdict === "fail" && reflectionCount > 0) { + log.warn( + "PIPELINE", + `Reflection retries exhausted (${reflectionCount}/${maxReflectionLoops}) — pipeline verdict still "fail", returning the original failed result` + ); + } + // ── Return result ───────────────────────────────────────────────────────── // Check if the last stage has a streaming Response const lastStage = result.stages[result.stages.length - 1]; diff --git a/open-sse/services/autoCombo/virtualFactory.ts b/open-sse/services/autoCombo/virtualFactory.ts index f468f23005f..2eff8257ea1 100644 --- a/open-sse/services/autoCombo/virtualFactory.ts +++ b/open-sse/services/autoCombo/virtualFactory.ts @@ -20,6 +20,7 @@ import { type AutoCategory, type AutoTier, } from "./suffixComposition"; +import { classifyTier } from "../tierResolver"; import type { AutoVariant } from "./autoPrefix"; import { buildFamilyCandidateFilter, type ModelFamily } from "./modelFamily"; import { getHiddenModelsByProvider } from "@/models"; @@ -602,6 +603,61 @@ export async function prepareVirtualAutoComboInputs( }; } +/** + * Score candidates at snapshot time using available data (capabilities, tier) + * and the mode-pack's dominant factors. Runtime telemetry (p95 latency, quota + * remaining) is not available during combo creation — this uses static signals only. + * + * Returns a map from modelStr → normalized weight score [0, 1]. + */ +export function computeSnapshotWeights( + candidates: readonly VirtualAutoComboCandidate[], + weights: ScoringWeights +): Map { + const scores = new Map(); + for (const c of candidates) { + let score = 0; + + // taskFit: reasoning + vision capable models score higher when taskFit is weighted + if (weights.taskFit > 0) { + if (c.resolvedReasoning || c.resolvedSupportsThinking) score += weights.taskFit * 0.6; + if (c.resolvedSupportsVision) score += weights.taskFit * 0.3; + } + + // stability: models with richer capabilities are assumed more stable + if (weights.stability > 0) { + const capabilityCount = + Number(c.resolvedReasoning ?? false) + + Number(c.resolvedSupportsThinking ?? false) + + Number(c.resolvedSupportsVision ?? false); + score += weights.stability * Math.min(capabilityCount / 2, 1); + } + + // Tier-based scoring (single classifyTier call covers both checks) + let tierInfo: { tier: string } | null = null; + if (weights.tierPriority > 0 || weights.costInv > 0) { + try { + tierInfo = classifyTier(c.provider, c.model); + } catch { + // fall through with zero + } + } + if (tierInfo && weights.tierPriority > 0 && tierInfo.tier === "premium") + score += weights.tierPriority; + if (tierInfo && weights.costInv > 0 && tierInfo.tier === "free") score += weights.costInv; + + // latencyInv: all candidates get a base score when latency matters + // (no runtime data at snapshot time, so equal baseline) + if (weights.latencyInv > 0) score += weights.latencyInv * 0.5; + + // health + quota: no runtime telemetry at snapshot time → neutral baseline + score += (weights.health + weights.quota) * 0.5; + + scores.set(c.modelStr, Math.min(score, 1)); + } + return scores; +} + function clonePreparedCandidates( candidates: readonly VirtualAutoComboCandidate[] ): VirtualAutoComboCandidate[] { @@ -770,6 +826,7 @@ export async function createVirtualAutoComboFromPrepared( } const providerPool = [...new Set(effectivePool.map((c) => c.provider))]; + const snapshotScores = computeSnapshotWeights(effectivePool, weights); const models = effectivePool.map((candidate, index) => ({ id: `virtual-auto-${variant || "default"}-${index + 1}-${candidate.provider}`, kind: "model" as const, @@ -779,7 +836,7 @@ export async function createVirtualAutoComboFromPrepared( ...(candidate.allowedConnectionIds ? { allowedConnectionIds: candidate.allowedConnectionIds } : {}), - weight: 1, + weight: snapshotScores.get(candidate.modelStr) ?? 1, label: candidate.provider, })); const autoConfig = { diff --git a/open-sse/services/autoRefreshDaemon.ts b/open-sse/services/autoRefreshDaemon.ts index 120b0815450..3a177a87ae6 100644 --- a/open-sse/services/autoRefreshDaemon.ts +++ b/open-sse/services/autoRefreshDaemon.ts @@ -125,8 +125,13 @@ class AutoRefreshDaemon { `[AutoRefreshDaemon] Credential expired for "${providerId}" (${config.displayName})` ); } - } catch { - // Network errors are non-fatal — retry next cycle + } catch (err) { + // Network errors are non-fatal — retry next cycle. G8: log which + // provider failed so credential problems are not silently masked. + console.warn( + `[AutoRefreshDaemon] Network error validating credential for "${providerId}" — retry next cycle`, + err instanceof Error ? err.message : err + ); } } @@ -165,8 +170,16 @@ class AutoRefreshDaemon { } return true; - } catch { - // Network errors (timeout, DNS failure) don't mean the credential is bad + } catch (err) { + // Network errors (timeout, DNS failure) don't mean the credential is bad. + // G8 (silent-stop fix): the previous bare `catch { return true; }` swallowed + // the error entirely — operators could never tell a credential was failing + // to validate due to network trouble. Log it (provider + reason) before + // returning the fail-open result. + console.warn( + `[AutoRefreshDaemon] Network error validating credential for "${providerId}" — treated as valid (fail-open), will retry next cycle`, + err instanceof Error ? err.message : err + ); return true; } finally { clearTimeout(timeout); diff --git a/open-sse/services/batchProcessor.ts b/open-sse/services/batchProcessor.ts index 578814427a0..e9fe915afd8 100644 --- a/open-sse/services/batchProcessor.ts +++ b/open-sse/services/batchProcessor.ts @@ -506,14 +506,46 @@ async function processSingleItemWithRetry(item: BatchRequestItem, apiKey: string } } +// G10 (silent-stop fix): individual batch-item dispatches can hang indefinitely +// if the upstream route stalls (no signal/timeout plumbed through). Bound each +// item with a wall-clock timeout so a stuck item fails fast (recorded as an item +// error) instead of freezing the whole batch loop. The orphaned dispatch keeps +// running in the background but can no longer block the batch. +export const BATCH_ITEM_DISPATCH_TIMEOUT_MS = 120_000; + +/** + * G10: race a promise against a wall-clock deadline. Exported for unit testing + * (batch dispatch is a module-internal import, so the timeout mechanism itself + * is verified directly here). + */ +export function withItemDispatchTimeout( + promise: Promise, + timeoutMs: number, + label: string +): Promise { + let timer: ReturnType | undefined; + const timeoutPromise = new Promise((_, reject) => { + timer = setTimeout( + () => reject(new Error(`${label} timed out after ${timeoutMs}ms`)), + timeoutMs + ); + }); + return Promise.race([promise, timeoutPromise]).finally(() => { + if (timer) clearTimeout(timer); + }); +} + async function processSingleItem(item: BatchRequestItem, apiKey: string) { const body = buildRequestBody(item); - - return await dispatch.dispatchBatchApiRequest({ - endpoint: item.url, - body, - apiKey, - }); + return withItemDispatchTimeout( + dispatch.dispatchBatchApiRequest({ + endpoint: item.url, + body, + apiKey, + }), + BATCH_ITEM_DISPATCH_TIMEOUT_MS, + `Batch item dispatch (${item.url})` + ); } export function buildRequestBody(item: BatchRequestItem) { diff --git a/open-sse/services/chatgptTlsClient.ts b/open-sse/services/chatgptTlsClient.ts index 57e5365726f..fd7b3f45509 100644 --- a/open-sse/services/chatgptTlsClient.ts +++ b/open-sse/services/chatgptTlsClient.ts @@ -1,629 +1,48 @@ /** * Browser-TLS-impersonating HTTP client for chatgpt.com. * - * Why this exists: ChatGPT's Cloudflare config pins `cf_clearance` to the - * client's TLS fingerprint (JA3/JA4) + HTTP/2 SETTINGS frame ordering. - * Node's Undici fetch presents an obvious "not a browser" handshake and - * gets challenged with `cf-mitigated: challenge` — even with all the right - * cookies. This module wraps `tls-client-node` (native shared library - * built from bogdanfinn/tls-client) to send a Firefox handshake instead. - * - * The first call lazily starts the managed sidecar; subsequent calls reuse - * a singleton TLSClient. Process exit hooks stop the sidecar cleanly. + * Thin re-export over the shared `tlsClientBase.ts` factory + * (`createTlsClientModule`). All provider-agnostic logic (sidecar lifecycle, + * streaming tail-file, proxy resolution, error classes, SSE detection) lives + * in the base module; this file supplies only ChatGPT-specific config and + * preserves the original public export surface. */ -import { tmpdir } from "node:os"; -import { join } from "node:path"; -import { mkdtemp, open, unlink, rmdir, stat, readFile } from "node:fs/promises"; -import { randomUUID } from "node:crypto"; -import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; - -let clientPromise: Promise | null = null; -let exitHookInstalled = false; +import { + createTlsClientModule, + type TlsFetchOptions, + type TlsFetchResult, +} from "./tlsClientBase.ts"; -const CHATGPT_PROFILE = "firefox_148"; // matches the Firefox 148 UA we send const DEFAULT_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_CHATGPT_TLS_TIMEOUT_MS || "", 10) || 60_000; -// Grace period added to the binding's wire-level timeout before our JS-level -// hard timeout fires. Under healthy operation `tls-client-node` honors -// `timeoutMilliseconds` and rejects on its own; the JS-level race only wins -// when the koffi-loaded native library is wedged (which the binding's own -// timer can't escape). Keep the grace small so users don't wait noticeably -// longer than the configured timeout when the binding is dead. const HARD_TIMEOUT_GRACE_MS = Number.parseInt(process.env.OMNIROUTE_CHATGPT_TLS_GRACE_MS || "", 10) || 10_000; const STREAM_FIRST_BYTE_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_CHATGPT_STREAM_FIRST_BYTE_TIMEOUT_MS || "", 10) || 30_000; -function installExitHook(): void { - if (exitHookInstalled) return; - exitHookInstalled = true; - const stop = async () => { - if (!clientPromise) return; - try { - const c = (await clientPromise) as { stop?: () => Promise }; - await c.stop?.(); - } catch { - // ignore - } - }; - process.once("beforeExit", stop); - process.once("SIGINT", () => { - void stop(); - }); - process.once("SIGTERM", () => { - void stop(); - }); -} - -/** - * Drop the cached client so the next `getClient()` call respawns it. Called - * when a request observes the native binding has wedged — releasing the - * reference lets a fresh TLSClient (and a fresh koffi load) take over without - * a process restart. - */ -function resetClientCache(): void { - clientPromise = null; -} - -export class TlsClientHangError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientHangError"; - } -} - -/** - * Race a `client.request()` promise against (a) a JS-level hard timeout and - * (b) the caller's abort signal. The native binding's `timeoutMilliseconds` - * already covers the wire path; this guards the case where the koffi binding - * itself deadlocks (observed after sustained load), where neither the - * binding's own timer nor a post-call `signal.aborted` re-check can recover. - */ -async function raceWithTimeout( - promise: Promise, - timeoutMs: number, - signal: AbortSignal | null | undefined -): Promise { - let timer: ReturnType | null = null; - let abortListener: (() => void) | null = null; - try { - const racers: Promise[] = [ - promise, - new Promise((_, reject) => { - timer = setTimeout(() => { - reject( - new TlsClientHangError( - `tls-client-node call exceeded ${timeoutMs}ms — native binding likely deadlocked` - ) - ); - }, timeoutMs); - }), - ]; - if (signal) { - racers.push( - new Promise((_, reject) => { - if (signal.aborted) { - reject(makeAbortError(signal)); - return; - } - abortListener = () => reject(makeAbortError(signal)); - signal.addEventListener("abort", abortListener, { once: true }); - }) - ); - } - return await Promise.race(racers); - } finally { - if (timer) clearTimeout(timer); - if (signal && abortListener) signal.removeEventListener("abort", abortListener); - } -} - -async function getClient(): Promise<{ - request: (url: string, opts: Record) => Promise; -}> { - if (!clientPromise) { - clientPromise = (async () => { - try { - const mod = await import("tls-client-node"); - const TLSClient = (mod as { TLSClient: new (opts?: Record) => unknown }) - .TLSClient; - // Native mode loads the shared library directly via koffi, avoiding the - // managed sidecar's localhost HTTP calls that OmniRoute's global fetch - // proxy patch interferes with. - const client = new TLSClient(buildNativeTlsClientOptions()) as { - start: () => Promise; - request: (url: string, opts: Record) => Promise; - }; - await client.start(); - - installExitHook(); - return client; - } catch (err) { - clientPromise = null; - const msg = err instanceof Error ? err.message : String(err); - throw new TlsClientUnavailableError( - `TLS impersonation client failed to start: ${msg}. ` + - `Verify tls-client-node is installed and its native binary downloaded.` - ); - } - })(); - } - return clientPromise as Promise<{ - request: (url: string, opts: Record) => Promise; - }>; -} - -interface TlsResponseLike { - status: number; - headers: Record; - body: string; // for non-streaming requests, the full response body - cookies?: Record; - text: () => Promise; - bytes: () => Promise; - json: () => Promise; -} - -export class TlsClientUnavailableError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientUnavailableError"; - } -} - -export interface TlsFetchOptions { - method?: "GET" | "POST" | "PUT" | "PATCH" | "DELETE"; - headers?: Record; - body?: string; - timeoutMs?: number; - signal?: AbortSignal | null; - /** - * If true, the response body is streamed to a temp file and exposed as a - * ReadableStream. Use for SSE responses (the conversation - * endpoint). Otherwise, the full body is read into memory. - */ - stream?: boolean; - /** EOF marker the upstream sends to signal end of stream (default: "[DONE]"). */ - streamEofSymbol?: string; - /** - * If true, instructs the underlying tls-client to return the response body - * as a base64 `data:;base64,...` string (so binary payloads survive - * the JSON marshalling step). Required for image / binary downloads — - * without it, raw bytes get UTF-8-decoded and any non-ASCII byte is - * mangled. Default false (text mode). - */ - byteResponse?: boolean; - /** - * Optional upstream proxy URL (`http://user:pass@host:port` or - * `socks5://...`). When set, the request is tunneled through this proxy - * before reaching chatgpt.com. Required for hosts whose bare IP is - * flagged by ChatGPT/Cloudflare (Russia, datacenter ranges, etc.) — - * without it, every call leaks the host IP and gets edge-rejected with - * a templated 401 / `Invalid session cookie`. - * - * Resolution order: - * 1. `options.proxyUrl` (per-call override from caller) - * 2. `process.env.OMNIROUTE_TLS_PROXY_URL` (single-flag opt-in) - * 3. `process.env.HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (POSIX-standard fallback) - * - * The native `tls-client-node` binding does **not** consult Go's - * `http.ProxyFromEnvironment`, so the env vars need to be plumbed in - * here at the JS layer. The dashboard's global-fetch monkey-patch only - * reaches Node's undici, not the koffi-loaded shared library used here. - */ - proxyUrl?: string; -} - -import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; -import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; - -/** - * Resolve the proxy URL for a tls-client request. Per-call value wins; - * otherwise we use the standard proxy fetch resolution which reads from - * the dashboard AsyncLocalStorage context or falls back to env vars. - * - * Fail-closed: if resolution throws (e.g. a configured socks5 proxy with - * ENABLE_SOCKS5_PROXY=false), this rethrows rather than returning undefined — - * undefined would let the native binding connect directly and leak the real IP. - */ -function resolveProxyUrl(perCall: string | undefined): string | undefined { - return resolveTlsClientProxyUrl("https://chatgpt.com", perCall, resolveProxyForRequest); -} - -export interface TlsFetchResult { - status: number; - headers: Headers; - /** Full response body as text — only populated for non-streaming requests. */ - text: string | null; - /** Streaming body — only populated when options.stream === true. */ - body: ReadableStream | null; -} - -// Test-only injection point. Tests call __setTlsFetchOverrideForTesting() -// to replace the real TLS client with a mock; production never touches this. -let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = - null; - -export function __setTlsFetchOverrideForTesting(fn: typeof testOverride): void { - testOverride = fn; -} - -/** - * Make a single HTTP request to chatgpt.com with a Firefox-like TLS fingerprint. - * - * Throws TlsClientUnavailableError if the native binary failed to load. - */ -export async function tlsFetchChatGpt( +export const tlsClientModule = createTlsClientModule({ + providerName: "ChatGPT", + tlsProfile: "firefox_148", + domain: "https://chatgpt.com", + tempDirPrefix: "cgpt-stream-", + tailFileVariant: "A", + responseValidation: "sse", + exportCloudflareCheck: false, + exposeStreamingForTesting: true, + defaultTimeoutMs: DEFAULT_TIMEOUT_MS, + hardTimeoutGraceMs: HARD_TIMEOUT_GRACE_MS, + firstByteTimeoutMs: STREAM_FIRST_BYTE_TIMEOUT_MS, +}); + +export const tlsFetchChatGpt = ( url: string, options: TlsFetchOptions = {} -): Promise { - if (testOverride) return testOverride(url, options); - // Honor abort signals up-front. tls-client-node's koffi binding doesn't - // accept an AbortSignal mid-flight (the binary call is opaque), so the best - // we can do is bail before issuing the call. We also re-check after — if - // the caller aborted while the upstream was running, throw rather than - // returning a stale response so the caller doesn't try to use it. - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - const client = await getClient(); - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - - const requestOptions: Record = { - method: options.method || "GET", - headers: options.headers || {}, - body: options.body, - tlsClientIdentifier: CHATGPT_PROFILE, - timeoutMilliseconds: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, - followRedirects: true, - withRandomTLSExtensionOrder: true, - isByteResponse: options.byteResponse === true, - // Plumb the configured proxy through to the native binding. tls-client-node - // consults `proxyUrl` in the per-call options (it does NOT auto-pick up - // HTTP_PROXY / HTTPS_PROXY env), so callers / env have to be threaded in - // explicitly. See `resolveProxyUrl()` for the lookup order. Without this - // line, every chatgpt-web call egresses with the bare host IP regardless - // of dashboard proxy config — see #2022. - proxyUrl: resolveProxyUrl(options.proxyUrl), - }; - - if (options.stream) { - return await tlsFetchStreaming( - client, - url, - requestOptions, - options.streamEofSymbol, - options.signal ?? null, - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS, - STREAM_FIRST_BYTE_TIMEOUT_MS - ); - } - - let tlsResponse: TlsResponseLike; - try { - tlsResponse = await raceWithTimeout( - client.request(url, requestOptions), - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS, - options.signal ?? null - ); - } catch (err) { - if (err instanceof TlsClientHangError) { - // The native binding is wedged — drop the singleton so the next - // request respawns a fresh client (and a fresh koffi load). - resetClientCache(); - } - throw err; - } - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - return { - status: tlsResponse.status, - headers: toHeaders(tlsResponse.headers), - text: tlsResponse.body, - body: null, - }; -} - -function makeAbortError(signal: AbortSignal): Error { - const reason = signal.reason; - if (reason instanceof Error) return reason; - const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); - err.name = "AbortError"; - return err; -} - -function toHeaders(raw: Record): Headers { - const h = new Headers(); - for (const [k, vs] of Object.entries(raw || {})) { - for (const v of vs) h.append(k, v); - } - return h; -} - -// ─── Streaming via temp file ──────────────────────────────────────────────── -// tls-client-node's streaming primitive writes the response body chunk-by-chunk -// to a file path, terminating when the upstream sends `streamOutputEOFSymbol`. -// We tail the file from a worker and surface the bytes as a ReadableStream. - -async function tlsFetchStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS, - firstByteTimeoutMs: number = STREAM_FIRST_BYTE_TIMEOUT_MS -): Promise { - const dir = await mkdtemp(join(tmpdir(), "cgpt-stream-")); - const path = join(dir, `${randomUUID()}.sse`); - - const streamOpts = { - ...requestOptions, - streamOutputPath: path, - streamOutputBlockSize: 1024, - streamOutputEOFSymbol: eofSymbol, - }; - - // Kick off the request without awaiting — tls-client writes the body to - // `path` chunk-by-chunk while the call runs. The Promise resolves when the - // request fully completes (full body written). Wrapping in raceWithTimeout - // guarantees this promise eventually settles even if the koffi binding - // wedges; on hang we reset the singleton so the next request respawns. - let resetOnHang = true; - const requestPromise = raceWithTimeout( - client.request(url, streamOpts), - hardTimeoutMs, - signal - ).catch((err: unknown) => { - if (resetOnHang && err instanceof TlsClientHangError) { - resetClientCache(); - resetOnHang = false; - } - // Re-throw so downstream consumers (waitForContent, tailFile) observe - // the rejection and surface it instead of treating the stream as having - // ended cleanly. - throw err; - }); - - // Wait for the file to exist AND have at least one byte. tls-client-node - // creates the output file when the request starts, but the file can be - // empty for a brief window before the first body chunk lands — peeking - // during that window would return "" and misclassify the response as - // non-SSE, dropping us into the buffered-wait branch and silently turning - // a streaming request into a buffered one. Waiting for content avoids - // that race; if the request actually fails before producing any bytes, - // the timeout falls through to the requestPromise drain below (returning - // the real upstream status). - const ready = await waitForContent(path, firstByteTimeoutMs, requestPromise); - if (!ready) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - // If the first byte arrived after our first-byte wait but before the - // request settled, tls-client-node may have written the full SSE body to - // streamOutputPath while leaving r.body empty. Prefer those captured bytes - // over misclassifying a successful delayed stream as "empty response body". - const fileText = await readTextFileIfExists(path); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: fileText || r.body, - body: null, - }; - } - - // Peek the first bytes to decide whether this looks like SSE. Anything - // that doesn't positively look like SSE (JSON `{...}`, HTML `<...>`, plain - // text rate-limit messages, Cloudflare challenge pages, etc.) gets surfaced - // as a non-streaming response so the executor sees the real upstream status - // and body — otherwise non-2xx error pages get silently treated as 200 OK - // and the SSE parser produces an empty completion. - const peek = await readFirstBytes(path, 256); - if (!looksLikeSse(peek)) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - const fileText = await readTextFileIfExists(path); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body || fileText, - body: null, - }; - } - - // Looks like SSE — start tailing. SSE bodies in practice are always 2xx; - // tls-client-node doesn't expose response status separately from full-body - // completion, so we report 200 and let the SSE parser consume the stream. - const stream = tailFile(path, eofSymbol, requestPromise, signal); - const headers = new Headers({ - "Content-Type": "text/event-stream", - "Cache-Control": "no-cache", - }); - return { status: 200, headers, text: null, body: stream }; -} - -/** - * Returns true if the peeked response body looks like an SSE stream — i.e., - * begins (after any leading whitespace) with one of the SSE field markers - * (`data:`, `event:`, `id:`, `retry:`) or a comment line (`:`). - * - * Exported for tests. - */ -export function looksLikeSse(text: string): boolean { - const trimmed = text.replace(/^[\s\r\n]+/, ""); - if (!trimmed) return false; - if (trimmed.startsWith(":")) return true; - return /^(data|event|id|retry):/i.test(trimmed); -} - -async function cleanupTempPath(path: string): Promise { - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); -} - -async function readTextFileIfExists(path: string): Promise { - try { - return await readFile(path, "utf8"); - } catch { - return ""; - } -} - -export async function __tlsFetchStreamingForTesting( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS, - firstByteTimeoutMs: number = STREAM_FIRST_BYTE_TIMEOUT_MS -): Promise { - return tlsFetchStreaming( - client as { request: (url: string, opts: Record) => Promise }, - url, - requestOptions, - eofSymbol, - signal, - hardTimeoutMs, - firstByteTimeoutMs - ); -} - -async function readFirstBytes(path: string, n: number): Promise { - const fd = await open(path, "r"); - try { - const buf = Buffer.alloc(n); - const { bytesRead } = await fd.read(buf, 0, n, 0); - return buf.subarray(0, bytesRead).toString("utf8"); - } finally { - await fd.close().catch(() => {}); - } -} - -/** - * Wait for the streaming output file to exist AND contain at least one byte. - * Returns false if the request settles before any bytes arrive (so the caller - * can drain `requestPromise` and surface the real upstream status). Returns - * true as soon as the file has data — even one byte is enough for the SSE - * heuristic to give a useful answer. - */ -async function waitForContent( - path: string, - timeoutMs: number, - requestPromise: Promise -): Promise { - let requestSettled = false; - requestPromise.then( - () => { - requestSettled = true; - }, - () => { - requestSettled = true; - } - ); - const start = Date.now(); - while (Date.now() - start < timeoutMs) { - try { - const s = await stat(path); - if (s.size > 0) return true; - } catch { - // file doesn't exist yet - } - // If the request finished without producing any bytes, no point waiting - // out the rest of the timeout — let the caller drain it. - if (requestSettled) return false; - await sleep(25); - } - return false; -} - -function tailFile( - path: string, - eofSymbol: string, - done: Promise, - signal: AbortSignal | null = null -): ReadableStream { - return new ReadableStream({ - async start(controller) { - const fd = await open(path, "r"); - const buf = Buffer.alloc(64 * 1024); - let offset = 0; - let finished = false; - let aborted = false; - let upstreamError: Error | null = null; - - // Track request settlement, capturing both fulfillment and rejection. - // Without the rejection branch, a mid-stream tls-client-node error - // becomes an unhandledRejection — the stream cleans up silently and - // the consumer sees what looks like a successful truncated response. - done.then( - () => { - finished = true; - }, - (err) => { - upstreamError = err instanceof Error ? err : new Error(String(err)); - finished = true; - } - ); - - // If the caller aborts, stop tailing immediately. - const onAbort = () => { - aborted = true; - }; - if (signal) { - if (signal.aborted) aborted = true; - else signal.addEventListener("abort", onAbort, { once: true }); - } +): Promise => tlsClientModule.tlsFetch(url, options); +export const __tlsFetchStreamingForTesting = tlsClientModule.__tlsFetchStreamingForTesting; - let errored = false; - try { - while (!aborted) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offset); - if (bytesRead > 0) { - const chunk = buf.subarray(0, bytesRead); - offset += bytesRead; - const text = chunk.toString("utf8"); - if (text.includes(eofSymbol)) { - const cutAt = text.indexOf(eofSymbol) + eofSymbol.length; - controller.enqueue(new Uint8Array(chunk.subarray(0, cutAt))); - break; - } - controller.enqueue(new Uint8Array(chunk)); - } else if (finished) { - // No more data and request completed. If the request rejected, - // surface the error so the consumer doesn't think the stream - // ended cleanly. - if (upstreamError) { - controller.error(upstreamError); - errored = true; - } - break; - } else { - await sleep(25); - } - } - } catch (err) { - controller.error(err); - errored = true; - } finally { - if (signal) signal.removeEventListener("abort", onAbort); - await fd.close().catch(() => {}); - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); - if (!errored) controller.close(); - } - }, - }); -} +export const __setTlsFetchOverrideForTesting = tlsClientModule.__setTlsFetchOverrideForTesting; -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); -} +export { TlsClientHangError, TlsClientUnavailableError } from "./tlsClientBase.ts"; +export type { TlsFetchOptions, TlsFetchResult } from "./tlsClientBase.ts"; +export { looksLikeSse } from "./tlsClientBase.ts"; diff --git a/open-sse/services/claudeTlsClient.ts b/open-sse/services/claudeTlsClient.ts index eb57220da8e..4ab4746195b 100644 --- a/open-sse/services/claudeTlsClient.ts +++ b/open-sse/services/claudeTlsClient.ts @@ -1,617 +1,49 @@ /** * Browser-TLS-impersonating HTTP client for claude.ai. * - * Why this exists: Claude's Cloudflare config pins `cf_clearance` to the - * client's TLS fingerprint (JA3/JA4) + HTTP/2 SETTINGS frame ordering. - * Node's Undici fetch presents an obvious "not a browser" handshake and - * gets challenged with `cf-mitigated: challenge` — even with all the right - * cookies. This module wraps `tls-client-node` (native shared library - * built from bogdanfinn/tls-client) to send a Chrome handshake instead. - * - * The first call lazily starts the managed sidecar; subsequent calls reuse - * a singleton TLSClient. Process exit hooks stop the sidecar cleanly. + * Thin re-export over the shared `tlsClientBase.ts` factory + * (`createTlsClientModule`). All provider-agnostic logic (sidecar lifecycle, + * streaming tail-file, proxy resolution, error classes, SSE detection) lives + * in the base module; this file supplies only Claude-specific config and + * preserves the original public export surface. */ -import { tmpdir } from "node:os"; -import { join } from "node:path"; -import { mkdtemp, open, unlink, rmdir, stat } from "node:fs/promises"; -import { randomUUID } from "node:crypto"; -import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; - -let clientPromise: Promise | null = null; -let exitHookInstalled = false; +import { + createTlsClientModule, + type TlsFetchOptions, + type TlsFetchResult, +} from "./tlsClientBase.ts"; export const CLAUDE_TLS_BROWSER_MAJOR_VERSION = "146"; -const CLAUDE_PROFILE = `chrome_${CLAUDE_TLS_BROWSER_MAJOR_VERSION}`; + const DEFAULT_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_CLAUDE_TLS_TIMEOUT_MS || "", 10) || 60_000; -// Grace period added to the binding's wire-level timeout before our JS-level -// hard timeout fires. Under healthy operation `tls-client-node` honors -// `timeoutMilliseconds` and rejects on its own; the JS-level race only wins -// when the koffi-loaded native library is wedged (which the binding's own -// timer can't escape). Keep the grace small so users don't wait noticeably -// longer than the configured timeout when the binding is dead. const HARD_TIMEOUT_GRACE_MS = Number.parseInt(process.env.OMNIROUTE_CLAUDE_TLS_GRACE_MS || "", 10) || 10_000; -function installExitHook(): void { - if (exitHookInstalled) return; - exitHookInstalled = true; - const stop = async () => { - if (!clientPromise) return; - try { - const c = (await clientPromise) as { stop?: () => Promise }; - await c.stop?.(); - } catch { - // ignore - } - }; - process.once("beforeExit", stop); - process.once("SIGINT", () => { - void stop(); - }); - process.once("SIGTERM", () => { - void stop(); - }); -} - -/** - * Drop the cached client so the next `getClient()` call respawns it. Called - * when a request observes the native binding has wedged — releasing the - * reference lets a fresh TLSClient (and a fresh koffi load) take over without - * a process restart. - */ -function resetClientCache(): void { - clientPromise = null; -} - -export class TlsClientHangError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientHangError"; - } -} - -/** - * Race a `client.request()` promise against (a) a JS-level hard timeout and - * (b) the caller's abort signal. The native binding's `timeoutMilliseconds` - * already covers the wire path; this guards the case where the koffi binding - * itself deadlocks (observed after sustained load), where neither the - * binding's own timer nor a post-call `signal.aborted` re-check can recover. - */ -async function raceWithTimeout( - promise: Promise, - timeoutMs: number, - signal: AbortSignal | null | undefined -): Promise { - let timer: ReturnType | null = null; - let abortListener: (() => void) | null = null; - try { - const racers: Promise[] = [ - promise, - new Promise((_, reject) => { - timer = setTimeout(() => { - reject( - new TlsClientHangError( - `tls-client-node call exceeded ${timeoutMs}ms — native binding likely deadlocked` - ) - ); - }, timeoutMs); - }), - ]; - if (signal) { - racers.push( - new Promise((_, reject) => { - if (signal.aborted) { - reject(makeAbortError(signal)); - return; - } - abortListener = () => reject(makeAbortError(signal)); - signal.addEventListener("abort", abortListener, { once: true }); - }) - ); - } - return await Promise.race(racers); - } finally { - if (timer) clearTimeout(timer); - if (signal && abortListener) signal.removeEventListener("abort", abortListener); - } -} - -async function getClient(): Promise<{ - request: (url: string, opts: Record) => Promise; -}> { - if (!clientPromise) { - clientPromise = (async () => { - try { - const mod = await import("tls-client-node"); - const TLSClient = (mod as { TLSClient: new (opts?: Record) => unknown }) - .TLSClient; - // Native mode loads the shared library directly via koffi, avoiding the - // managed sidecar's localhost HTTP calls that OmniRoute's global fetch - // proxy patch interferes with. - const client = new TLSClient(buildNativeTlsClientOptions()) as { - start: () => Promise; - request: (url: string, opts: Record) => Promise; - }; - await client.start(); - - installExitHook(); - return client; - } catch (err) { - clientPromise = null; - const msg = err instanceof Error ? err.message : String(err); - throw new TlsClientUnavailableError( - `TLS impersonation client failed to start: ${msg}. ` + - `Verify tls-client-node is installed and its native binary downloaded.` - ); - } - })(); - } - return clientPromise as Promise<{ - request: (url: string, opts: Record) => Promise; - }>; -} - -interface TlsResponseLike { - status: number; - headers: Record; - body: string; // for non-streaming requests, the full response body - cookies?: Record; - text: () => Promise; - bytes: () => Promise; - json: () => Promise; -} - -export class TlsClientUnavailableError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientUnavailableError"; - } -} - -export interface TlsFetchOptions { - method?: "GET" | "POST" | "PUT" | "PATCH" | "DELETE"; - headers?: Record; - body?: string; - timeoutMs?: number; - signal?: AbortSignal | null; - /** - * If true, the response body is streamed to a temp file and exposed as a - * ReadableStream. Use for SSE responses (the conversation - * endpoint). Otherwise, the full body is read into memory. - */ - stream?: boolean; - /** EOF marker the upstream sends to signal end of stream (default: "[DONE]"). */ - streamEofSymbol?: string; - /** - * If true, instructs the underlying tls-client to return the response body - * as a base64 `data:;base64,...` string (so binary payloads survive - * the JSON marshalling step). Required for image / binary downloads — - * without it, raw bytes get UTF-8-decoded and any non-ASCII byte is - * mangled. Default false (text mode). - */ - byteResponse?: boolean; - /** - * Optional upstream proxy URL (`http://user:pass@host:port` or - * `socks5://...`). When set, the request is tunneled through this proxy - * before reaching claude.ai. Required for hosts whose bare IP is - * flagged by Claude/Cloudflare (Russia, datacenter ranges, etc.) — - * without it, every call leaks the host IP and gets edge-rejected with - * a templated 401 / `Invalid session cookie`. - * - * Resolution order: - * 1. `options.proxyUrl` (per-call override from caller) - * 2. `process.env.OMNIROUTE_TLS_PROXY_URL` (single-flag opt-in) - * 3. `process.env.HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (POSIX-standard fallback) - * - * The native `tls-client-node` binding does **not** consult Go's - * `http.ProxyFromEnvironment`, so the env vars need to be plumbed in - * here at the JS layer. The dashboard's global-fetch monkey-patch only - * reaches Node's undici, not the koffi-loaded shared library used here. - */ - proxyUrl?: string; -} - -import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; -import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; - -/** - * Resolve the proxy URL for a tls-client request. Per-call value wins; - * otherwise we use the standard proxy fetch resolution which reads from - * the dashboard AsyncLocalStorage context or falls back to env vars. - * - * Fail-closed: if resolution throws (e.g. a configured socks5 proxy with - * ENABLE_SOCKS5_PROXY=false), this rethrows rather than returning undefined — - * undefined would let the native binding connect directly and leak the real IP. - */ -function resolveProxyUrl(perCall: string | undefined): string | undefined { - return resolveTlsClientProxyUrl("https://claude.ai", perCall, resolveProxyForRequest); -} - -export interface TlsFetchResult { - status: number; - headers: Headers; - /** Full response body as text — only populated for non-streaming requests. */ - text: string | null; - /** Streaming body — only populated when options.stream === true. */ - body: ReadableStream | null; -} - -// Test-only injection point. Tests call __setTlsFetchOverrideForTesting() -// to replace the real TLS client with a mock; production never touches this. -let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = - null; - -export function __setTlsFetchOverrideForTesting(fn: typeof testOverride): void { - testOverride = fn; -} - -/** - * Make a single HTTP request to claude.ai with the configured Chrome TLS profile. - * - * Throws TlsClientUnavailableError if the native binary failed to load. - */ -export async function tlsFetchClaude( +export const tlsClientModule = createTlsClientModule({ + providerName: "Claude", + tlsProfile: `chrome_${CLAUDE_TLS_BROWSER_MAJOR_VERSION}`, + domain: "https://claude.ai", + tempDirPrefix: "cgpt-stream-", + tailFileVariant: "A", + responseValidation: "sse", + exportCloudflareCheck: false, + exposeStreamingForTesting: true, + // Claude waits indefinitely for the first SSE byte (original 2-arg waitForContent). + defaultTimeoutMs: DEFAULT_TIMEOUT_MS, + hardTimeoutGraceMs: HARD_TIMEOUT_GRACE_MS, + firstByteTimeoutMs: Number.POSITIVE_INFINITY, +}); + +export const tlsFetchClaude = ( url: string, options: TlsFetchOptions = {} -): Promise { - if (testOverride) return testOverride(url, options); - // Honor abort signals up-front. tls-client-node's koffi binding doesn't - // accept an AbortSignal mid-flight (the binary call is opaque), so the best - // we can do is bail before issuing the call. We also re-check after — if - // the caller aborted while the upstream was running, throw rather than - // returning a stale response so the caller doesn't try to use it. - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - const client = await getClient(); - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - - const requestOptions: Record = { - method: options.method || "GET", - headers: options.headers || {}, - body: options.body, - tlsClientIdentifier: CLAUDE_PROFILE, - timeoutMilliseconds: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, - followRedirects: true, - withRandomTLSExtensionOrder: true, - isByteResponse: options.byteResponse === true, - // Plumb the configured proxy through to the native binding. tls-client-node - // consults `proxyUrl` in the per-call options (it does NOT auto-pick up - // HTTP_PROXY / HTTPS_PROXY env), so callers / env have to be threaded in - // explicitly. See `resolveProxyUrl()` for the lookup order. Without this - // line, every chatgpt-web call egresses with the bare host IP regardless - // of dashboard proxy config — see #2022. - proxyUrl: resolveProxyUrl(options.proxyUrl), - }; - - if (options.stream) { - return await tlsFetchStreaming( - client, - url, - requestOptions, - options.streamEofSymbol, - options.signal ?? null, - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS - ); - } - - let tlsResponse: TlsResponseLike; - try { - tlsResponse = await raceWithTimeout( - client.request(url, requestOptions), - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS, - options.signal ?? null - ); - } catch (err) { - if (err instanceof TlsClientHangError) { - // The native binding is wedged — drop the singleton so the next - // request respawns a fresh client (and a fresh koffi load). - resetClientCache(); - } - throw err; - } - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - return { - status: tlsResponse.status, - headers: toHeaders(tlsResponse.headers), - text: tlsResponse.body, - body: null, - }; -} - -function makeAbortError(signal: AbortSignal): Error { - const reason = signal.reason; - if (reason instanceof Error) return reason; - const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); - err.name = "AbortError"; - return err; -} - -function toHeaders(raw: Record): Headers { - const h = new Headers(); - for (const [k, vs] of Object.entries(raw || {})) { - for (const v of vs) h.append(k, v); - } - return h; -} - -// ─── Streaming via temp file ──────────────────────────────────────────────── -// tls-client-node's streaming primitive writes the response body chunk-by-chunk -// to a file path, terminating when the upstream sends `streamOutputEOFSymbol`. -// We tail the file from a worker and surface the bytes as a ReadableStream. - -// Cap for the bounded fallback read of a non-SSE error body straight from the -// streaming temp file (mirrors the 2048-byte cap executors/claude-web.ts -// already applies when reading error bodies) — avoids buffering an unbounded -// error page into memory. See #7134. -const MAX_ERROR_BODY_BYTES = 16 * 1024; - -/** - * Exported for tests (issue #7134): allows injecting a fake `client` so the - * non-SSE error-body fallback path can be exercised without - * `--experimental-test-module-mocks`, matching the DI pattern already used - * by `__setTlsFetchOverrideForTesting` for the outer `tlsFetchClaude`. - */ -export async function tlsFetchStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS -): Promise { - const dir = await mkdtemp(join(tmpdir(), "cgpt-stream-")); - const path = join(dir, `${randomUUID()}.sse`); - - const streamOpts = { - ...requestOptions, - streamOutputPath: path, - streamOutputBlockSize: 1024, - streamOutputEOFSymbol: eofSymbol, - }; - - // Kick off the request without awaiting — tls-client writes the body to - // `path` chunk-by-chunk while the call runs. The Promise resolves when the - // request fully completes (full body written). Wrapping in raceWithTimeout - // guarantees this promise eventually settles even if the koffi binding - // wedges; on hang we reset the singleton so the next request respawns. - let resetOnHang = true; - const requestPromise = raceWithTimeout( - client.request(url, streamOpts), - hardTimeoutMs, - signal - ).catch((err: unknown) => { - if (resetOnHang && err instanceof TlsClientHangError) { - resetClientCache(); - resetOnHang = false; - } - // Re-throw so downstream consumers (waitForContent, tailFile) observe - // the rejection and surface it instead of treating the stream as having - // ended cleanly. - throw err; - }); - - // Wait for the file to exist AND have at least one byte. tls-client-node - // creates the output file when the request starts, but the file can be - // empty for a brief window before the first body chunk lands — peeking - // during that window would return "" and misclassify the response as - // non-SSE, dropping us into the buffered-wait branch and silently turning - // a streaming request into a buffered one. Waiting for content avoids - // that race; if the request actually fails before producing any bytes, - // the timeout falls through to the requestPromise drain below (returning - // the real upstream status). - // Do not impose a second, shorter first-byte timeout here. Opus-class - // models can legitimately take more than five seconds before emitting the - // first SSE event. `requestPromise` is already guarded by the configured - // wire timeout plus the JS hard-timeout grace, so waiting until either the - // file has data or that promise settles remains bounded. - const ready = await waitForContent(path, requestPromise); - if (!ready) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Peek the first bytes to decide whether this looks like SSE. Anything - // that doesn't positively look like SSE (JSON `{...}`, HTML `<...>`, plain - // text rate-limit messages, Cloudflare challenge pages, etc.) gets surfaced - // as a non-streaming response so the executor sees the real upstream status - // and body — otherwise non-2xx error pages get silently treated as 200 OK - // and the SSE parser produces an empty completion. - const peek = await readFirstBytes(path, 256); - if (!looksLikeSse(peek)) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - // tls-client-node's `streamOutputPath` mode writes the response body to - // the temp file chunk-by-chunk and does NOT also populate the resolved - // response's in-memory `body` field (confirmed against - // node_modules/tls-client-node/dist/response.js) — so for every non-SSE, - // non-2xx claude-web response (400/403/429/500 with a real JSON/HTML - // error), `r.body` is empty even though the real bytes are sitting in - // `path` (we just peeked them above). Prefer `r.body` when it IS - // populated (some native-client modes do fill it in); otherwise fall - // back to a bounded read of the temp file so the real upstream error - // detail reaches the caller instead of being silently discarded. #7134 - const text = r.body || (await readFirstBytes(path, MAX_ERROR_BODY_BYTES).catch(() => "")); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text, - body: null, - }; - } - - // Looks like SSE — start tailing. SSE bodies in practice are always 2xx; - // tls-client-node doesn't expose response status separately from full-body - // completion, so we report 200 and let the SSE parser consume the stream. - const stream = tailFile(path, eofSymbol, requestPromise, signal); - const headers = new Headers({ - "Content-Type": "text/event-stream", - "Cache-Control": "no-cache", - }); - return { status: 200, headers, text: null, body: stream }; -} - -/** - * Returns true if the peeked response body looks like an SSE stream — i.e., - * begins (after any leading whitespace) with one of the SSE field markers - * (`data:`, `event:`, `id:`, `retry:`) or a comment line (`:`). - * - * Exported for tests. - */ -export function looksLikeSse(text: string): boolean { - const trimmed = text.replace(/^[\s\r\n]+/, ""); - if (!trimmed) return false; - if (trimmed.startsWith(":")) return true; - return /^(data|event|id|retry):/i.test(trimmed); -} - -async function cleanupTempPath(path: string): Promise { - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); -} - -async function readFirstBytes(path: string, n: number): Promise { - const fd = await open(path, "r"); - try { - const buf = Buffer.alloc(n); - const { bytesRead } = await fd.read(buf, 0, n, 0); - return buf.subarray(0, bytesRead).toString("utf8"); - } finally { - await fd.close().catch(() => {}); - } -} - -/** - * Wait for the streaming output file to exist AND contain at least one byte. - * Returns false if the request settles before any bytes arrive (so the caller - * can drain `requestPromise` and surface the real upstream status). Returns - * true as soon as the file has data — even one byte is enough for the SSE - * heuristic to give a useful answer. - */ -async function waitForContent( - path: string, - requestPromise: Promise -): Promise { - let requestSettled = false; - requestPromise.then( - () => { - requestSettled = true; - }, - () => { - requestSettled = true; - } - ); - while (true) { - try { - const s = await stat(path); - if (s.size > 0) return true; - } catch { - // file doesn't exist yet - } - // If the request finished without producing any bytes, no point waiting - // out the rest of the timeout — let the caller drain it. - if (requestSettled) return false; - await sleep(25); - } -} - -function tailFile( - path: string, - eofSymbol: string, - done: Promise, - signal: AbortSignal | null = null -): ReadableStream { - return new ReadableStream({ - async start(controller) { - const fd = await open(path, "r"); - const buf = Buffer.alloc(64 * 1024); - let offset = 0; - let finished = false; - let aborted = false; - let upstreamError: Error | null = null; - - // Track request settlement, capturing both fulfillment and rejection. - // Without the rejection branch, a mid-stream tls-client-node error - // becomes an unhandledRejection — the stream cleans up silently and - // the consumer sees what looks like a successful truncated response. - done.then( - () => { - finished = true; - }, - (err) => { - upstreamError = err instanceof Error ? err : new Error(String(err)); - finished = true; - } - ); - - // If the caller aborts, stop tailing immediately. - const onAbort = () => { - aborted = true; - }; - if (signal) { - if (signal.aborted) aborted = true; - else signal.addEventListener("abort", onAbort, { once: true }); - } +): Promise => tlsClientModule.tlsFetch(url, options); +export const tlsFetchStreaming = tlsClientModule.__tlsFetchStreamingForTesting; - let errored = false; - try { - while (!aborted) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offset); - if (bytesRead > 0) { - const chunk = buf.subarray(0, bytesRead); - offset += bytesRead; - const text = chunk.toString("utf8"); - if (text.includes(eofSymbol)) { - const cutAt = text.indexOf(eofSymbol) + eofSymbol.length; - controller.enqueue(new Uint8Array(chunk.subarray(0, cutAt))); - break; - } - controller.enqueue(new Uint8Array(chunk)); - } else if (finished) { - // No more data and request completed. If the request rejected, - // surface the error so the consumer doesn't think the stream - // ended cleanly. - if (upstreamError) { - controller.error(upstreamError); - errored = true; - } - break; - } else { - await sleep(25); - } - } - } catch (err) { - controller.error(err); - errored = true; - } finally { - if (signal) signal.removeEventListener("abort", onAbort); - await fd.close().catch(() => {}); - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); - if (!errored) controller.close(); - } - }, - }); -} +export const __setTlsFetchOverrideForTesting = tlsClientModule.__setTlsFetchOverrideForTesting; -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); -} +export { TlsClientHangError, TlsClientUnavailableError } from "./tlsClientBase.ts"; +export type { TlsFetchOptions, TlsFetchResult } from "./tlsClientBase.ts"; +export { looksLikeSse } from "./tlsClientBase.ts"; diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 9b4069bbc00..f01d206ecab 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -46,6 +46,7 @@ import { getDefaultComboConfig, resolveComboQueueDepth, isComboCooldownWaitEligible, + resolveComboTargetTimeoutMsForCombo, } from "./comboConfig.ts"; import { maybeGenerateHandoff, @@ -82,6 +83,7 @@ import { normalizeStickinessMessages, recordStickyBinding, clearStickyBinding, + clearStickyBindingsForCombo, peekStickyConnectionId, resolveDisableSessionStickiness, } from "./combo/sessionStickiness.ts"; @@ -89,6 +91,7 @@ import { selectQuotaShareTarget } from "./combo/quotaShareStrategy.ts"; import { makeConnectionConcurrencyResolver, lookupPositiveCap } from "./combo/concurrencyCaps.ts"; import { acquireQuotaShareConcurrencySlot } from "./combo/quotaShareConcurrency.ts"; import { canAffordRequest } from "../../src/lib/quota/quotaScheduler.ts"; +import { resolveConnectionTimeoutMs } from "../handlers/chatCore/upstreamTimeouts.ts"; import { getCachedProviderConnectionById } from "../../src/lib/db/readCache.ts"; import { orderTargetsByEvalScores } from "./evalRouting.ts"; @@ -133,6 +136,7 @@ import { isProviderInCooldown, recordProviderCooldown } from "./providerCooldown import { resolveResilienceSettings, type ResilienceSettings, + type ComboCooldownWaitSettings, } from "../../src/lib/resilience/settings"; import { resolveReasoningBufferedMaxTokens, toPositiveInteger } from "./reasoningTokenBuffer.ts"; import { RESET_WINDOW_NAMES } from "./combo/types.ts"; @@ -141,6 +145,7 @@ import type { ComboRetryAfter, ComboErrorBody, SingleModelTarget, + ComboLogger, HandleComboChatOptions, HandleRoundRobinOptions, ResolvedComboTarget, @@ -179,6 +184,8 @@ import { TRANSIENT_FOR_SEMAPHORE, MAX_FALLBACK_WAIT_MS, MAX_GLOBAL_ATTEMPTS, + COMBO_LOOP_SAFETY_TIMEOUT_MS, + COMBO_SAFETY_DRAIN_MS, isAllAccountsRateLimitedResponse, clampComboDepth, shouldSkipForPredictedTtft, @@ -619,6 +626,40 @@ export { pinIsDurablyUnhealthy }; /** @param {string} errorText */ /** @param {object} options */ +/** + * Resolves the per-target timeout ceiling for a combo target: when the target's + * connection carries `providerSpecificData.timeoutMs`, re-runs + * resolveComboTargetTimeoutMsForCombo with that timeout as the ceiling so the + * combo's per-target timer follows the selected connection. + * Returns undefined when the connection or its timeout is absent — the runner + * then falls back to the setup-time comboTargetTimeoutMs. + */ +export async function resolveTargetTimeoutMsForTarget( + config: Record | null | undefined, + strategy: string, + comboCooldownWait: Pick, + target?: SingleModelTarget, + log?: Pick | null +): Promise { + const connectionId = target && "connectionId" in target ? target.connectionId : null; + if (!connectionId) return undefined; + try { + const connection = await getCachedProviderConnectionById(connectionId); + if (!connection) return undefined; + const timeoutMs = resolveConnectionTimeoutMs(connection.providerSpecificData); + if (timeoutMs === undefined) return undefined; + return resolveComboTargetTimeoutMsForCombo(config, timeoutMs, strategy, comboCooldownWait); + } catch (err) { + log?.debug?.( + "COMBO", + `resolveTargetTimeoutMsForTarget connection lookup failed: ${ + err instanceof Error ? err.message : String(err) + }` + ); + return undefined; + } +} + /** * #10681 egress: every combo response carries the opaque trace id in an * `X-OmniRoute-Combo-Trace` header so a post-incident lookup of the ordered @@ -681,6 +722,14 @@ async function handleComboChatInner({ const handleSingleModelWithTimeout = buildTargetTimeoutRunner({ handleSingleModel, comboTargetTimeoutMs, + resolveTargetTimeoutMs: (target) => + resolveTargetTimeoutMsForTarget( + config, + strategy, + resilienceSettings.comboCooldownWait, + target, + log + ), log, }); @@ -1046,11 +1095,54 @@ async function handleComboChatInner({ const globalPromise = new Promise((res) => { globalResolve = res; }); + + // G1 (silent-stop fix): the speculative loop's `Promise.race` waits on + // `globalPromise`, which is ONLY resolved from inside a task (success or + // fatal error). If a target hangs — e.g. the operator disabled the per-model + // timeout (`targetTimeoutMs: 0`) and the upstream never settles — the race + // never resolves and the request hangs forever with no response. This safety + // promise force-resolves after the combo budget (comboTimeoutMs when set, + // otherwise a hard ceiling) so the request ALWAYS terminates with an + // actionable 504 instead of dying silently. `comboExpired` is flipped so the + // target loop stops launching new work; the existing comboExpired branch + // returns the aggregated 504. + const loopSafetyMs = + comboTimeoutMs > 0 ? comboTimeoutMs : COMBO_LOOP_SAFETY_TIMEOUT_MS; + let loopSafetyFired = false; + let loopSafetyTimer: ReturnType | null = null; + const loopSafetyPromise = new Promise((resolve) => { + loopSafetyTimer = setTimeout(() => { + loopSafetyFired = true; + log.warn( + "COMBO", + `Combo loop safety timeout (${loopSafetyMs}ms) reached without a terminal response — force-terminating` + ); + resolve( + errorResponseWithComboDiagnostics( + 504, + `Combo global timeout (${loopSafetyMs}ms) without a terminal response`, + buildComboDiag("combo_timeout"), + { code: "COMBO_TIMEOUT", type: "server_error" } + ) + ); + }, loopSafetyMs); + loopSafetyTimer.unref?.(); + }); const runningTasks = new Set>(); let anySuccess = false; // #10681: steps already recorded as dispatched (so per-target retries do not // duplicate the decision). const dispatchedTargets = new Set(); + // G1: flip comboExpired as soon as the safety timer fires so the next loop + // iteration breaks instead of launching more targets after the budget, and + // abort every in-flight target so a hung upstream actually gets cancelled + // (not just "response stops"). + const markLoopExpiredIfSafetyFired = () => { + if (loopSafetyFired) { + comboExpired = true; + for (const [, ac] of abortControllers.entries()) ac.abort(); + } + }; const abortControllers = new Map(); const zeroLatencyOptimizationsEnabled = config.zeroLatencyOptimizationsEnabled === true; const hasProtectedPriorityTarget = @@ -2363,6 +2455,17 @@ async function handleComboChatInner({ })().catch((err) => { const logError = log.error ?? log.warn; logError("COMBO", `Speculative task error for target ${i}`, err); + // G2 (silent-stop fix): never leave the speculative loop waiting on an + // unresolved globalPromise. If a task throws unexpectedly (outside + // executeTarget's error handling) and no other task succeeds, the post-loop + // `Promise.race([globalPromise, ...])` would hang forever. Resolve with a + // 502 so the request terminates with an actionable error. + if (!anySuccess && globalResolve) { + anySuccess = true; + globalResolve( + errorResponse(502, `Combo target ${i} failed with an unexpected error`) + ); + } }); runningTasks.add(task); @@ -2380,10 +2483,11 @@ async function handleComboChatInner({ timeoutResolve = r; setTimeout(r, hedgeDelay); }); - await Promise.race([task, globalPromise, timeoutPromise]); + await Promise.race([task, globalPromise, timeoutPromise, loopSafetyPromise]); } else { - await Promise.race([task, globalPromise]); + await Promise.race([task, globalPromise, loopSafetyPromise]); } + markLoopExpiredIfSafetyFired(); // Global combo timeout check: after each target completes, stop trying // further targets if the total elapsed time exceeds comboTimeoutMs. @@ -2398,13 +2502,51 @@ async function handleComboChatInner({ } if (!anySuccess && runningTasks.size > 0) { - await Promise.race([globalPromise, Promise.all([...runningTasks])]); + // G1: include loopSafetyPromise so a hung last task (per-model timeout + // disabled) cannot freeze this post-loop race forever. + await Promise.race([globalPromise, Promise.all([...runningTasks]), loopSafetyPromise]); + markLoopExpiredIfSafetyFired(); + } + + // G1: if the safety timer won the race (request would otherwise hang), give + // in-flight tasks a short drain window to land their per-model errors into + // comboErrors so the 504 carries the same "tried: a (500)" summary the + // regular comboExpired branch produces — then return the safety 504. + if (loopSafetyFired && !anySuccess) { + if (runningTasks.size > 0) { + await Promise.race([ + Promise.allSettled([...runningTasks]), + new Promise((resolve) => setTimeout(resolve, COMBO_SAFETY_DRAIN_MS)), + ]); + } + const summary = comboErrors + .slice(0, 5) + .map((e) => `${e.model} (${e.status})`) + .join(", "); + const msg = + `Combo global timeout (${loopSafetyMs}ms) after ${recordedAttempts}/${orderedTargets.length} targets` + + (comboErrors.length > 0 + ? ` | tried: ${summary}${comboErrors.length > 5 ? `... (+${comboErrors.length - 5})` : ""}` + : "") + + " without a terminal response"; + return errorResponseWithComboDiagnostics( + 504, + msg, + buildComboDiag("combo_timeout"), + { code: "COMBO_TIMEOUT", type: "server_error" } + ); } // #10681: finalize the decision trace (success). finalizeComboTrace(traceInvocationId, orderedTargets); finishComboTrace(traceInvocationId, { status: 200 }); if (anySuccess) { + // G1: clear the safety timer on the happy path so a successful combo does + // not leave a 10-minute timer alive per request. + if (loopSafetyTimer) { + clearTimeout(loopSafetyTimer); + loopSafetyTimer = null; + } return await globalPromise; } @@ -2872,6 +3014,9 @@ async function handleRoundRobinCombo({ filteredTargets = await expandPromptCacheAffinityTargets(filteredTargets); modelCount = filteredTargets.length; } + if (disableSessionStickiness) { + clearStickyBindingsForCombo(combo.name); + } const _rrSessionSticky = disableSessionStickiness ? ({ targets: filteredTargets, messageHash: null, stuck: false } as const) : await applySessionStickiness( @@ -2923,6 +3068,33 @@ async function handleRoundRobinCombo({ // and the "Done with this model" path below), mirroring handleComboChat. const rrOutcomes: Array = []; + // G4 (silent-stop fix): round-robin has NO global timeout — a hung model + // (per-model timeout disabled via targetTimeoutMs: 0) would freeze the request + // forever with no response. Safety promise + timer bound the whole loop; when + // it fires, rrExpired flips and every subsequent model attempt short-circuits + // to the 504. Cleaned up in the loop's finally. + const rrConfiguredTimeoutMs = + (config as { comboTimeoutMs?: number }).comboTimeoutMs ?? 0; + const rrLoopSafetyMs = + rrConfiguredTimeoutMs > 0 ? rrConfiguredTimeoutMs : COMBO_LOOP_SAFETY_TIMEOUT_MS; + let rrExpired = false; + let rrLoopSafetyTimer: ReturnType | null = null; + let rrResolveSafety: ((res: Response) => void) | null = null; + const rrSafetyPromise = new Promise((resolve) => { + rrResolveSafety = resolve; + }); + rrLoopSafetyTimer = setTimeout(() => { + rrExpired = true; + log.warn( + "COMBO-RR", + `Round-robin loop exceeded ${rrLoopSafetyMs}ms without a terminal response — force-terminating` + ); + rrResolveSafety?.( + errorResponse(504, `Round-robin combo exceeded ${rrLoopSafetyMs}ms without a terminal response`) + ); + }, rrLoopSafetyMs); + rrLoopSafetyTimer.unref?.(); + // #1731: Per-request in-memory set of providers whose quota is fully exhausted. // When a target returns a quota-exhausted 429, remaining targets from the same // provider are skipped to avoid the cascade through N same-provider targets. @@ -2931,8 +3103,11 @@ async function handleRoundRobinCombo({ const transientRateLimitedProviders = new Set(); // Try each model starting from the round-robin target - for (let offset = 0; offset < modelCount; offset++) { - const modelIndex = (rrStartIndex + offset) % modelCount; + try { + for (let offset = 0; offset < modelCount; offset++) { + // G4: stop launching new work once the safety timer fired. + if (rrExpired) break; + const modelIndex = (rrStartIndex + offset) % modelCount; const target = filteredTargets[modelIndex]; const modelStr = target.modelStr; const provider = target.provider; @@ -3077,11 +3252,15 @@ async function handleRoundRobinCombo({ fingerprint: resolveTargetFingerprint(target) ?? "", }); - const result = await handleSingleModel(attemptBody, modelStr, { - ...targetForAttempt, - effectiveComboStrategy: "round-robin", - failoverBeforeRetry: config.failoverBeforeRetry, - }); + const result = await Promise.race([ + handleSingleModel(attemptBody, modelStr, { + ...targetForAttempt, + effectiveComboStrategy: "round-robin", + failoverBeforeRetry: config.failoverBeforeRetry, + }), + rrSafetyPromise, + ]); + if (rrExpired) return result; // G4: safety timer won — stop everything // Quota-aware scheduling: reserve the estimated budget for this // dispatch (opt-in, same env gate as the pre-request check). Best-effort @@ -3519,6 +3698,26 @@ async function handleRoundRobinCombo({ release(); } } + } catch (err) { + // G4: unexpected exception in the round-robin loop must never crash the + // request silently — surface a 500 instead of hanging the client. + log.error?.("COMBO-RR", "Unexpected error in round-robin loop", err); + return errorResponse(500, "Unexpected error in round-robin combo"); + } finally { + if (rrLoopSafetyTimer) { + clearTimeout(rrLoopSafetyTimer); + rrLoopSafetyTimer = null; + } + } + + // G4: if the safety timer fired between iterations (no race captured it), + // terminate with the actionable 504 instead of the generic exhaustion path. + if (rrExpired) { + return errorResponse( + 504, + `Round-robin combo exceeded ${rrLoopSafetyMs}ms without a terminal response` + ); + } // All models exhausted const latencyMs = Date.now() - startTime; diff --git a/open-sse/services/combo/comboPredicates.ts b/open-sse/services/combo/comboPredicates.ts index dc875092365..f7cfa3322d7 100644 --- a/open-sse/services/combo/comboPredicates.ts +++ b/open-sse/services/combo/comboPredicates.ts @@ -18,6 +18,15 @@ import type { ResolvedComboTarget } from "./types.ts"; // Status codes that should mark round-robin target semaphores as cooling down. export const TRANSIENT_FOR_SEMAPHORE = [429, 502, 503, 504]; +// G1 (silent-stop fix): hard ceiling for the combo target loop when the operator +// left comboTimeoutMs at 0 ("unlimited"). Without this, a hung upstream (per-model +// timeout disabled) would freeze the request forever with no response. 10 minutes +// is a generous bound for legitimate long-running fallback cascades. +export const COMBO_LOOP_SAFETY_TIMEOUT_MS = 10 * 60 * 1000; +// G1: after the safety timer fires, wait this long for in-flight targets to land +// their per-model errors into comboErrors (so the 504 carries the same "tried:" +// summary as the regular timeout path) before returning the safety response. +export const COMBO_SAFETY_DRAIN_MS = 2000; // Patterns that signal all accounts for a provider are rate-limited / exhausted. // Used to detect 503 responses from handleNoCredentials so combo can fallback. export const ALL_ACCOUNTS_RATE_LIMITED_PATTERNS = [ diff --git a/open-sse/services/combo/sessionStickiness.ts b/open-sse/services/combo/sessionStickiness.ts index 6f51f69df0c..7347730a602 100644 --- a/open-sse/services/combo/sessionStickiness.ts +++ b/open-sse/services/combo/sessionStickiness.ts @@ -77,6 +77,8 @@ interface StickyEntry { connectionId: string; createdAt: number; lastUsedAt: number; + /** Combo identity that owns this binding (matches `scopeMessageHash` namespace). */ + namespace?: string; } /** @@ -357,17 +359,23 @@ function evict(): void { } /** Record (or refresh) a sticky binding after a successful request. */ -export function recordStickyBinding(messageHash: string, connectionId: string): void { +export function recordStickyBinding( + messageHash: string, + connectionId: string, + namespace?: string +): void { const existing = stickyMap.get(messageHash); if (existing) { existing.connectionId = connectionId; existing.lastUsedAt = Date.now(); + if (namespace) existing.namespace = namespace; } else { evict(); stickyMap.set(messageHash, { connectionId, createdAt: Date.now(), lastUsedAt: Date.now(), + ...(namespace ? { namespace } : {}), }); } } @@ -377,6 +385,24 @@ export function clearStickyBinding(messageHash: string): void { stickyMap.delete(messageHash); } +/** + * Evict every in-memory sticky binding owned by a combo. + * + * Stale pins survive combo edits: `updateCombo` clears the persisted + * `session_model_history` rows, but the process-global sticky map is only + * bounded by TTL (15 min) — a binding recorded before the operator disabled + * stickiness or reordered models keeps promoting the old connection to + * position 0 for the remainder of the TTL window, silently defeating the + * combo's declared priority order (#XXXX). Combo writes call this so a + * config/model change takes effect immediately instead of after TTL expiry. + */ +export function clearStickyBindingsForCombo(namespace: string): void { + if (!namespace) return; + for (const [key, entry] of stickyMap) { + if (entry.namespace === namespace) stickyMap.delete(key); + } +} + /** * Read-only peek at the connectionId currently bound to `messageHash`, without * mutating the store or checking TTL/health. Lets combo.ts's failure paths @@ -462,6 +488,10 @@ export async function applySessionStickiness( const existing = stickyMap.get(messageHash); if (!existing) return { targets: orderedTargets, messageHash, stuck: false }; + // Backfill the owning namespace so combo-scoped eviction (combo edit / + // stickiness disable) can find bindings recorded before this field existed. + if (namespace && existing.namespace !== namespace) existing.namespace = namespace; + // Check TTL if (Date.now() - existing.lastUsedAt > TTL_MS) { stickyMap.delete(messageHash); diff --git a/open-sse/services/combo/targetResolution.ts b/open-sse/services/combo/targetResolution.ts index ff3b0362f7d..eeec177c86a 100644 --- a/open-sse/services/combo/targetResolution.ts +++ b/open-sse/services/combo/targetResolution.ts @@ -72,6 +72,7 @@ import { } from "./rrState.ts"; import { applySessionStickiness, + clearStickyBindingsForCombo, normalizeStickinessMessages, resolveDisableSessionStickiness, type ApplyStickinessResult, @@ -458,6 +459,15 @@ async function applyContinuityFilters( config as Record | null | undefined, settings as Record | null | undefined ); + // Evict any in-memory sticky bindings this combo still owns when stickiness is + // disabled. Disabling stops NEW bindings, but a binding recorded while it was + // enabled would otherwise keep re-promoting the old connection for the rest of + // the 15-minute TTL — silently defeating the combo's priority order until the + // binding ages out or the process restarts (user report: disabling stickiness + // on orchestrator still pinned opencode-go/mimo-v2.5-max first). + if (disableSessionStickiness) { + clearStickyBindingsForCombo(combo.name); + } const sticky: ApplyStickinessResult = disableSessionStickiness ? { targets: initialOrderedTargets, messageHash: null, stuck: false } : await applySessionStickiness( diff --git a/open-sse/services/combo/targetTimeoutRunner.ts b/open-sse/services/combo/targetTimeoutRunner.ts index 4eb6cb8b684..402d093f354 100644 --- a/open-sse/services/combo/targetTimeoutRunner.ts +++ b/open-sse/services/combo/targetTimeoutRunner.ts @@ -93,19 +93,34 @@ export function buildTargetTimeoutRunner(deps: { handleSingleModel: HandleSingleModel; comboTargetTimeoutMs: number; log: ComboLogger; + resolveTargetTimeoutMs?: ( + target?: SingleModelTarget + ) => Promise | number | undefined; }): ( b: Record, modelStr: string, target?: SingleModelTarget ) => Promise { - const { handleSingleModel, comboTargetTimeoutMs, log } = deps; + const { handleSingleModel, comboTargetTimeoutMs, log, resolveTargetTimeoutMs } = deps; ensureDiagnosticListener(); return async ( b: Record, modelStr: string, target?: SingleModelTarget ): Promise => { - if (comboTargetTimeoutMs <= 0) { + const resolvedTimeoutMs = await resolveTargetTimeoutMs?.(target); + const effectiveTimeoutMs = + typeof resolvedTimeoutMs === "number" && Number.isFinite(resolvedTimeoutMs) + ? resolvedTimeoutMs + : comboTargetTimeoutMs; + if (effectiveTimeoutMs <= 0) { + // G3 (silent-stop fix): a disabled per-model timeout means a hung upstream + // stalls the target until the combo loop safety timer (COMBO_LOOP_SAFETY_TIMEOUT_MS) + // force-terminates — surface that dependency instead of silently running bare. + log.warn( + "COMBO", + `Per-model combo timeout is DISABLED (effectiveTimeoutMs=${effectiveTimeoutMs}) for ${modelStr} — a hung upstream will hang this target until the combo loop safety timeout` + ); return handleSingleModel(b, modelStr, target).catch((err) => errorResponse(502, err?.message ?? "Upstream model error") ); @@ -120,13 +135,13 @@ export function buildTargetTimeoutRunner(deps: { const abortErr = new Error(COMBO_PER_MODEL_TIMEOUT_REASON); recordTimeoutContext({ modelStr, - timeoutMs: comboTargetTimeoutMs, + timeoutMs: effectiveTimeoutMs, abortError: abortErr, timestamp: Date.now(), }); log.warn( "COMBO", - `Model ${modelStr} exceeded ${comboTargetTimeoutMs}ms timeout — falling back` + `Model ${modelStr} exceeded ${effectiveTimeoutMs}ms timeout — falling back` ); timeoutController.abort(abortErr); // HTTP 504 (not proprietary 524): this is OmniRoute's own per-target timer. @@ -147,7 +162,7 @@ export function buildTargetTimeoutRunner(deps: { } ) ); - }, comboTargetTimeoutMs); + }, effectiveTimeoutMs); }); const targetWithSignal = { ...(target ?? {}), diff --git a/open-sse/services/compression/engines/cavemanAdapter.ts b/open-sse/services/compression/engines/cavemanAdapter.ts index 0fa8e5bb9e4..d07e0b0c91f 100644 --- a/open-sse/services/compression/engines/cavemanAdapter.ts +++ b/open-sse/services/compression/engines/cavemanAdapter.ts @@ -262,6 +262,10 @@ export const liteEngine: CompressionEngine = { }, apply(body, options) { const adapter = adaptBodyForCompression(body); + // stepConfig is Record, so its compressToolResults is `unknown`. + // Only an explicit boolean counts as a step override — anything else falls through + // to global config.lite, then the default (keeps the type `boolean`, and a malformed + // step value can no longer leak through the `??` chain as `{}`). const stepCompressToolResults = options?.stepConfig?.compressToolResults; const result = applyLiteCompression(adapter.body, { ...options, diff --git a/open-sse/services/compression/engines/headroom/gcf/decode_generic.ts b/open-sse/services/compression/engines/headroom/gcf/decode_generic.ts index 9c9a7774d98..f9597e7b2ef 100644 --- a/open-sse/services/compression/engines/headroom/gcf/decode_generic.ts +++ b/open-sse/services/compression/engines/headroom/gcf/decode_generic.ts @@ -1,7 +1,8 @@ /** * GCF generic-profile decoder (decodeGeneric). * Vendored from gcf-typescript — generic profile only. Current with GCF spec v3.2 - * (nested object flattening) and the [N]: inline-array quoting fix. + * (nested object flattening), the [N]: inline-array quoting fix, the int64/2^53 numeric- + * domain rendering (SPEC 2.3.1), and the root-array surplus count check (SPEC 13). * https://github.com/blackwell-systems/gcf-typescript * * SPDX-License-Identifier: MIT @@ -78,7 +79,14 @@ export function decodeGeneric(input: string): any { // Root array. if (first.startsWith("## [")) { - const [arr] = parseArrayFromHeader(contentLines, 0, 0, first.slice(3)); + const [arr, consumed] = parseArrayFromHeader(contentLines, 0, 0, first.slice(3)); + // A root array spans the whole document, so any structural line past the consumed + // rows is a surplus item, not sibling content. The row loop stops at the declared + // count, so the count assert only catches the deficit; surplus is caught here (SPEC + // Section 13: a mismatch, fewer OR more items than declared, is an error). + if (consumed < contentLines.length) { + throw new Error("count_mismatch: declared count is fewer than the rows present"); + } return arr; } diff --git a/open-sse/services/compression/engines/headroom/gcf/index.ts b/open-sse/services/compression/engines/headroom/gcf/index.ts index 5671ced952a..6be512f61e1 100644 --- a/open-sse/services/compression/engines/headroom/gcf/index.ts +++ b/open-sse/services/compression/engines/headroom/gcf/index.ts @@ -1,7 +1,8 @@ /** * GCF (Graph Compact Format) — generic profile encoder/decoder. * Vendored from gcf-typescript for zero-dependency integration. Current with - * GCF spec v3.2 (nested object flattening) + [N]: inline-array quoting fix. + * GCF spec v3.2 (nested object flattening) + [N]: inline-array quoting fix + int64/2^53 + * numeric-domain rendering (SPEC 2.3.1) + root-array surplus count check (SPEC 13). * https://github.com/blackwell-systems/gcf-typescript * * SPDX-License-Identifier: MIT diff --git a/open-sse/services/compression/engines/headroom/gcf/scalar.ts b/open-sse/services/compression/engines/headroom/gcf/scalar.ts index f7e82d44191..f3b5082362a 100644 --- a/open-sse/services/compression/engines/headroom/gcf/scalar.ts +++ b/open-sse/services/compression/engines/headroom/gcf/scalar.ts @@ -1,7 +1,8 @@ /** * Common scalar grammar for GCF (Graph Compact Format). * Vendored from gcf-typescript — generic profile only. Current with GCF spec v3.2 - * (nested object flattening) and the [N]: inline-array quoting fix. + * (nested object flattening), the [N]: inline-array quoting fix, the int64/2^53 numeric- + * domain rendering (SPEC 2.3.1), and the root-array surplus count check (SPEC 13). * https://github.com/blackwell-systems/gcf-typescript * * SPDX-License-Identifier: MIT @@ -107,7 +108,12 @@ export function formatNumber(f: number): string { if (Object.is(f, -0)) return "-0"; if (f === 0) return "0"; const abs = Math.abs(f); - if (abs >= 1e-6 && abs < 1e21) { + // Plain decimal only below 2^53. Every double at or above 2^53 is integer-valued, so a + // plain rendering emits a bare-integer token: indistinguishable from an int64 on the wire + // and beyond a JavaScript decoder's safe-integer range (2^53-1), so it is rejected/misread + // on decode. Exponent shape keeps bare tokens int64 and decimal/exponent tokens doubles + // (SPEC 2.3.1). 2^53 = 9007199254740992. + if (abs >= 1e-6 && abs < 9007199254740992) { return toPreciseDecimal(f); } // Exponent notation. diff --git a/open-sse/services/cursorApiKeyAuth.ts b/open-sse/services/cursorApiKeyAuth.ts index 60c3385783d..126123b119d 100644 --- a/open-sse/services/cursorApiKeyAuth.ts +++ b/open-sse/services/cursorApiKeyAuth.ts @@ -50,8 +50,12 @@ export function isCursorApiKey(value: unknown): value is string { return typeof value === "string" && value.startsWith(CURSOR_API_KEY_PREFIX); } +// Session-cache key fingerprint, not a password/credential hash — keyed with a fixed context +// label so it reads as a domain-separated digest rather than a bare password hash. function cacheKeyFor(apiKey: string): string { - return crypto.createHash("sha256").update(apiKey).digest("hex"); + return crypto.createHmac("sha256", "omniroute-cursor-session-cache-fingerprint-v1") + .update(apiKey) + .digest("hex"); } export function readJwtExpiryMs(token: string): number | null { diff --git a/open-sse/services/defaultReasoningEffort.ts b/open-sse/services/defaultReasoningEffort.ts index 156312a24c3..4c09643147a 100644 --- a/open-sse/services/defaultReasoningEffort.ts +++ b/open-sse/services/defaultReasoningEffort.ts @@ -30,18 +30,28 @@ function hasExplicitReasoningField(body: Record): boolean { * * `suffixEffort` (#7694) is the tier a `/-{effort}` synced-model alias * resolved to (`src/sse/services/model.ts`'s `resolveSyncedModelIdAndEffort`) — an - * explicit, request-time model selection, so it takes priority over the static - * `ModelSpec.defaultReasoningEffort` fleet-wide default (#6879) when both are present. + * explicit, request-time model selection, so it takes priority over both defaults + * below when present. + * + * `syncedDefaultEffort` is the vendor-declared default captured at sync time + * (`reasoning.default_effort`, e.g. OpenRouter `stealth/ox-alpha` declares `max`) — + * see `detectDefaultThinkingEffort`. A model that only produces usable output with + * an explicit effort gets the vendor default instead of an empty upstream response. + * It is the LOWEST-priority default: an explicit client value wins, the suffix alias + * wins, and a static `ModelSpec.defaultReasoningEffort` (operator-configured + * strip-by-default, #6879) also wins over the vendor default. */ export function applyDefaultReasoningEffort>( body: T, modelId: string, - suffixEffort?: string | null + suffixEffort?: string | null, + syncedDefaultEffort?: string | null ): T { if (!body || typeof body !== "object") return body; if (hasExplicitReasoningField(body)) return body; - const defaultEffort = suffixEffort || getModelSpec(modelId)?.defaultReasoningEffort; + const defaultEffort = + suffixEffort || getModelSpec(modelId)?.defaultReasoningEffort || syncedDefaultEffort; if (!defaultEffort) return body; return { ...body, reasoning_effort: defaultEffort }; diff --git a/open-sse/services/errorClassifier.ts b/open-sse/services/errorClassifier.ts index 735f44eb089..2776644de1f 100644 --- a/open-sse/services/errorClassifier.ts +++ b/open-sse/services/errorClassifier.ts @@ -28,13 +28,17 @@ export function isEmptyContentResponse(responseBody: unknown): boolean { const content = message?.content ?? delta?.content; const reasoningContent = message?.reasoning_content ?? delta?.reasoning_content; + // opencode-routed gateways (e.g. opencode/mimo-v2.5-free) name the reasoning + // field `reasoning` instead of `reasoning_content` (#6623). + const reasoningAlt = message?.reasoning ?? delta?.reasoning; const hasToolCalls = (Array.isArray(message?.tool_calls) && (message.tool_calls as unknown[]).length > 0) || (Array.isArray(delta?.tool_calls) && (delta.tool_calls as unknown[]).length > 0); const hasContent = content !== null && content !== undefined && content !== ""; const hasReasoning = - reasoningContent !== null && reasoningContent !== undefined && reasoningContent !== ""; + (reasoningContent !== null && reasoningContent !== undefined && reasoningContent !== "") || + (reasoningAlt !== null && reasoningAlt !== undefined && reasoningAlt !== ""); // A response truncated at the token limit (finish_reason "length") is a valid, // successful completion even with empty text — do not flag it as a fake success. diff --git a/open-sse/services/grokTlsClient.ts b/open-sse/services/grokTlsClient.ts index 685f3710805..00a952dd70d 100644 --- a/open-sse/services/grokTlsClient.ts +++ b/open-sse/services/grokTlsClient.ts @@ -1,608 +1,41 @@ /** * Browser-TLS-impersonating HTTP client for grok.com. * - * Why this exists: Grok sits behind Cloudflare Enterprise which pins - * `cf_clearance` to the client's TLS fingerprint (JA3/JA4) + HTTP/2 SETTINGS - * frame ordering. Node's Undici fetch presents an obvious "not a browser" - * handshake and gets challenged with a 403 "Request rejected by anti-bot - * rules." — even with a valid `sso` + `sso-rw` session cookie. This module - * wraps `tls-client-node` (native shared library built from - * bogdanfinn/tls-client) to send a Chrome handshake instead. - * - * Mirrors `perplexityTlsClient.ts`; kept as an independent module so changes - * here cannot regress the production chatgpt-web / perplexity-web paths. - * The first call lazily starts the managed sidecar; subsequent calls reuse - * a singleton TLSClient. Process exit hooks stop the sidecar cleanly. - * - * Issue: #3180 + * Thin re-export over the shared `tlsClientBase.ts` factory + * (`createTlsClientModule`). All provider-agnostic logic (sidecar lifecycle, + * streaming tail-file, proxy resolution, error classes, Cloudflare challenge + * detection) lives in the base module; this file supplies only Grok-specific + * config and preserves the original public export surface. */ -import { tmpdir } from "node:os"; -import { join, dirname } from "node:path"; -import { mkdtemp, open, unlink, rmdir, stat } from "node:fs/promises"; -import { randomUUID } from "node:crypto"; -import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; +import { + createTlsClientModule, + type TlsFetchOptions, + type TlsFetchResult, +} from "./tlsClientBase.ts"; -let clientPromise: Promise | null = null; -let exitHookInstalled = false; - -const GROK_PROFILE = "chrome_146"; // closest supported wreq-js profile (chrome_149 absent in 2.3.1, #5591) const DEFAULT_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_GROK_TLS_TIMEOUT_MS || "", 10) || 60_000; -// Grace period added to the binding's wire-level timeout before our JS-level -// hard timeout fires. Under healthy operation `tls-client-node` honors -// `timeoutMilliseconds` and rejects on its own; the JS-level race only wins -// when the koffi-loaded native library is wedged (which the binding's own -// timer can't escape). Keep the grace small so users don't wait noticeably -// longer than the configured timeout when the binding is dead. const HARD_TIMEOUT_GRACE_MS = Number.parseInt(process.env.OMNIROUTE_GROK_TLS_GRACE_MS || "", 10) || 10_000; -function installExitHook(): void { - if (exitHookInstalled) return; - exitHookInstalled = true; - const stop = async () => { - if (clientPromise === null) return; - try { - const c = (await clientPromise) as { stop?: () => Promise }; - await c.stop?.(); - } catch { - // ignore - } - }; - process.once("beforeExit", stop); - process.once("SIGINT", () => { - void stop(); - }); - process.once("SIGTERM", () => { - void stop(); - }); -} - -/** - * Drop the cached client so the next `getClient()` call respawns it. Called - * when a request observes the native binding has wedged — releasing the - * reference lets a fresh TLSClient (and a fresh koffi load) take over without - * a process restart. - */ -function resetClientCache(): void { - clientPromise = null; -} - -export class TlsClientHangError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientHangError"; - } -} - -/** - * Race a `client.request()` promise against (a) a JS-level hard timeout and - * (b) the caller's abort signal. The native binding's `timeoutMilliseconds` - * already covers the wire path; this guards the case where the koffi binding - * itself deadlocks (observed after sustained load), where neither the - * binding's own timer nor a post-call `signal.aborted` re-check can recover. - */ -async function raceWithTimeout( - promise: Promise, - timeoutMs: number, - signal: AbortSignal | null | undefined -): Promise { - let timer: ReturnType | null = null; - let abortListener: (() => void) | null = null; - try { - const racers: Promise[] = [ - promise, - new Promise((_, reject) => { - timer = setTimeout(() => { - reject( - new TlsClientHangError( - `tls-client-node call exceeded ${timeoutMs}ms — native binding likely deadlocked` - ) - ); - }, timeoutMs); - }), - ]; - if (signal) { - racers.push( - new Promise((_, reject) => { - if (signal.aborted) { - reject(makeAbortError(signal)); - return; - } - abortListener = () => reject(makeAbortError(signal)); - signal.addEventListener("abort", abortListener, { once: true }); - }) - ); - } - return await Promise.race(racers); - } finally { - if (timer) clearTimeout(timer); - if (signal && abortListener) signal.removeEventListener("abort", abortListener); - } -} - -async function getClient(): Promise<{ - request: (url: string, opts: Record) => Promise; -}> { - if (!clientPromise) { - clientPromise = (async () => { - try { - const mod = await import("tls-client-node"); - const TLSClient = (mod as { TLSClient: new (opts?: Record) => unknown }) - .TLSClient; - // Native mode loads the shared library directly via koffi, avoiding the - // managed sidecar's localhost HTTP calls that OmniRoute's global fetch - // proxy patch interferes with. - const client = new TLSClient(buildNativeTlsClientOptions()) as { - start: () => Promise; - request: (url: string, opts: Record) => Promise; - }; - await client.start(); - - installExitHook(); - return client; - } catch (err) { - clientPromise = null; - const msg = err instanceof Error ? err.message : String(err); - throw new TlsClientUnavailableError( - `TLS impersonation client failed to start: ${msg}. ` + - `Verify tls-client-node is installed and its native binary downloaded.` - ); - } - })(); - } - return clientPromise as Promise<{ - request: (url: string, opts: Record) => Promise; - }>; -} - -interface TlsResponseLike { - status: number; - headers: Record; - body: string; // for non-streaming requests, the full response body - cookies?: Record; - text: () => Promise; - bytes: () => Promise; - json: () => Promise; -} - -export class TlsClientUnavailableError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientUnavailableError"; - } -} - -export interface TlsFetchOptions { - method?: "GET" | "POST" | "PUT" | "PATCH" | "DELETE"; - headers?: Record; - body?: string; - timeoutMs?: number; - signal?: AbortSignal | null; - /** - * If true, the response body is streamed to a temp file and exposed as a - * ReadableStream. Use for NDJSON streaming responses (the - * Grok conversation endpoint). Otherwise, the full body is read into memory. - */ - stream?: boolean; - /** EOF marker the upstream sends to signal end of stream (default: "[DONE]"). */ - streamEofSymbol?: string; - /** - * Optional upstream proxy URL (`http://user:pass@host:port` or - * `socks5://...`). When set, the request is tunneled through this proxy - * before reaching grok.com. - * - * Resolution order: - * 1. `options.proxyUrl` (per-call override from caller) - * 2. `process.env.OMNIROUTE_TLS_PROXY_URL` (single-flag opt-in) - * 3. `process.env.HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (POSIX-standard fallback) - * - * The native `tls-client-node` binding does **not** consult Go's - * `http.ProxyFromEnvironment`, so the env vars need to be plumbed in here at - * the JS layer. - */ - proxyUrl?: string; -} - -import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; -import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; - -/** - * Resolve the proxy URL for a tls-client request. Per-call value wins; - * otherwise we use the standard proxy fetch resolution which reads from - * the dashboard AsyncLocalStorage context or falls back to env vars. - * - * Fail-closed: if resolution throws (e.g. a configured socks5 proxy with - * ENABLE_SOCKS5_PROXY=false), this rethrows rather than returning undefined — - * undefined would let the native binding connect directly and leak the real IP. - */ -function resolveProxyUrl(perCall: string | undefined): string | undefined { - return resolveTlsClientProxyUrl("https://grok.com", perCall, resolveProxyForRequest); -} - -export interface TlsFetchResult { - status: number; - headers: Headers; - /** Full response body as text — only populated for non-streaming requests. */ - text: string | null; - /** Streaming body — only populated when options.stream === true. */ - body: ReadableStream | null; -} - -// Test-only injection point. Tests call __setTlsFetchOverrideForTesting() -// to replace the real TLS client with a mock; production never touches this. -let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = - null; - -export function __setTlsFetchOverrideForTesting(fn: typeof testOverride): void { - testOverride = fn; -} - -/** - * Make a single HTTP request to grok.com with a Chrome-like TLS fingerprint. - * - * Throws TlsClientUnavailableError if the native binary failed to load. - */ -export async function tlsFetchGrok( - url: string, - options: TlsFetchOptions = {} -): Promise { - if (testOverride) return testOverride(url, options); - // Honor abort signals up-front. tls-client-node's koffi binding doesn't - // accept an AbortSignal mid-flight (the binary call is opaque), so the best - // we can do is bail before issuing the call. We also re-check after — if - // the caller aborted while the upstream was running, throw rather than - // returning a stale response so the caller doesn't try to use it. - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - const client = await getClient(); - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - - const requestOptions: Record = { - method: options.method || "GET", - headers: options.headers || {}, - body: options.body, - tlsClientIdentifier: GROK_PROFILE, - timeoutMilliseconds: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, - followRedirects: true, - withRandomTLSExtensionOrder: true, - // Plumb the configured proxy through to the native binding. tls-client-node - // consults `proxyUrl` in the per-call options (it does NOT auto-pick up - // HTTP_PROXY / HTTPS_PROXY env), so callers / env have to be threaded in - // explicitly. See `resolveProxyUrl()` for the lookup order. - proxyUrl: resolveProxyUrl(options.proxyUrl), - }; - - if (options.stream) { - return await tlsFetchStreaming( - client, - url, - requestOptions, - options.streamEofSymbol, - options.signal ?? null, - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS - ); - } - - let tlsResponse: TlsResponseLike; - try { - tlsResponse = await raceWithTimeout( - client.request(url, requestOptions), - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS, - options.signal ?? null - ); - } catch (err) { - if (err instanceof TlsClientHangError) { - // The native binding is wedged — drop the singleton so the next - // request respawns a fresh client (and a fresh koffi load). - resetClientCache(); - } - throw err; - } - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - return { - status: tlsResponse.status, - headers: toHeaders(tlsResponse.headers), - text: tlsResponse.body, - body: null, - }; -} - -function makeAbortError(signal: AbortSignal): Error { - const reason = signal.reason; - if (reason instanceof Error) return reason; - const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); - err.name = "AbortError"; - return err; -} - -function toHeaders(raw: Record): Headers { - const h = new Headers(); - for (const [k, vs] of Object.entries(raw || {})) { - for (const v of vs) h.append(k, v); - } - return h; -} - -/** - * Returns true if the response body is a Cloudflare challenge/interstitial page - * rather than a real Grok response. From VPS/datacenter IPs a valid cookie - * still gets a 403 "Request rejected by anti-bot rules." JSON; distinguishing - * it from a genuine auth failure lets the caller surface an actionable error - * (issue #3180). - * - * Exported so the executor and the connection validator share one detector. - */ -export function isCloudflareChallenge(text: string | null | undefined): boolean { - if (!text) return false; - return /just a moment|window\._cf_chl_opt|challenges\.cloudflare\.com|attention required|cf-chl/i.test( - text - ); -} - -// ─── Streaming via temp file ──────────────────────────────────────────────── -// tls-client-node's streaming primitive writes the response body chunk-by-chunk -// to a file path, terminating when the upstream sends `streamOutputEOFSymbol`. -// We tail the file from a worker and surface the bytes as a ReadableStream. - -async function tlsFetchStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS -): Promise { - const dir = await mkdtemp(join(tmpdir(), "grok-stream-")); - const path = join(dir, `${randomUUID()}.ndjson`); - - const streamOpts = { - ...requestOptions, - streamOutputPath: path, - streamOutputBlockSize: 1024, - streamOutputEOFSymbol: eofSymbol, - }; - - // Kick off the request without awaiting — tls-client writes the body to - // `path` chunk-by-chunk while the call runs. The Promise resolves when the - // request fully completes (full body written). Wrapping in raceWithTimeout - // guarantees this promise eventually settles even if the koffi binding - // wedges; on hang we reset the singleton so the next request respawns. - let resetOnHang = true; - const requestPromise = raceWithTimeout( - client.request(url, streamOpts), - hardTimeoutMs, - signal - ).catch((err: unknown) => { - if (resetOnHang && err instanceof TlsClientHangError) { - resetClientCache(); - resetOnHang = false; - } - // Re-throw so downstream consumers (waitForContent, tailFile) observe - // the rejection and surface it instead of treating the stream as having - // ended cleanly. - throw err; - }); - - // Wait for the file to exist AND have at least one byte. - const ready = await waitForContent(path, 5_000, requestPromise); - if (!ready) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Peek at the first bytes to distinguish a genuine NDJSON stream from a - // Cloudflare challenge page or an HTML error response that tls-client-node - // streamed to the temp file with a 200 status. - const peek = await readFirstBytes(path, 256); - if (isCloudflareChallenge(peek)) { - await cleanupTempPath(path); - return { - status: 403, - headers: new Headers({ "Content-Type": "text/html" }), - text: peek, - body: null, - }; - } - if (peek.trimStart().startsWith("<")) { - // HTML error page (not a challenge) — surface as a non-2xx so the executor - // can emit a proper SSE error chunk instead of feeding HTML to the NDJSON - // parser. - await cleanupTempPath(path); - return { - status: 502, - headers: new Headers({ "Content-Type": "text/html" }), - text: peek, - body: null, - }; - } - - // Looks like NDJSON — start tailing. The requestPromise will eventually - // resolve with the real upstream status; tailFile propagates non-2xx errors - // into the stream so the consumer sees them instead of a truncated success. - const stream = tailFile(path, eofSymbol, requestPromise, signal); - const headers = new Headers({ - "Content-Type": "application/x-ndjson", - "Cache-Control": "no-cache", - }); - return { status: 200, headers, text: null, body: stream }; -} - -async function cleanupTempPath(path: string): Promise { - await unlink(path).catch(() => {}); - await rmdir(dirname(path)).catch(() => {}); -} - -async function readFirstBytes(path: string, n: number): Promise { - const fd = await open(path, "r"); - try { - const buf = Buffer.alloc(n); - const { bytesRead } = await fd.read(buf, 0, n, 0); - return buf.subarray(0, bytesRead).toString("utf8"); - } finally { - await fd.close().catch(() => {}); - } -} - -/** - * Wait for the streaming output file to exist AND contain at least one byte. - * Returns false if the request settles before any bytes arrive (so the caller - * can drain `requestPromise` and surface the real upstream status). Returns - * true as soon as the file has data — even one byte is enough for the NDJSON - * heuristic to give a useful answer. - */ -async function waitForContent( - path: string, - timeoutMs: number, - requestPromise: Promise -): Promise { - let requestSettled = false; - requestPromise.then( - () => { - requestSettled = true; - }, - () => { - requestSettled = true; - } - ); - const start = Date.now(); - while (Date.now() - start < timeoutMs) { - try { - const s = await stat(path); - if (s.size > 0) return true; - } catch { - // file doesn't exist yet - } - // If the request finished without producing any bytes, no point waiting - // out the rest of the timeout — let the caller drain it. - if (requestSettled) return false; - await sleep(25); - } - return false; -} - -function tailFile( - path: string, - eofSymbol: string, - done: Promise, - signal: AbortSignal | null = null -): ReadableStream { - return new ReadableStream({ - async start(controller) { - const fd = await open(path, "r"); - const buf = Buffer.alloc(64 * 1024); - let offset = 0; - let finished = false; - let aborted = false; - let upstreamError: Error | null = null; - - // Track request settlement, capturing both fulfillment and rejection. - // Without the rejection branch, a mid-stream tls-client-node error - // becomes an unhandledRejection — the stream cleans up silently and - // the consumer sees what looks like a successful truncated response. - done.then( - () => { - finished = true; - }, - (err) => { - upstreamError = err instanceof Error ? err : new Error(String(err)); - finished = true; - } - ); - - // If the caller aborts, stop tailing immediately. - const onAbort = () => { - aborted = true; - }; - if (signal) { - if (signal.aborted) aborted = true; - else signal.addEventListener("abort", onAbort, { once: true }); - } - - let errored = false; - try { - while (!aborted) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offset); - if (bytesRead > 0) { - const chunk = buf.subarray(0, bytesRead); - offset += bytesRead; - const text = chunk.toString("utf8"); - - // Check for EOF symbol in the chunk. - if (text.includes(eofSymbol)) { - const beforeEof = text.substring(0, text.indexOf(eofSymbol)); - if (beforeEof) { - controller.enqueue(Buffer.from(beforeEof, "utf8")); - } - controller.close(); - return; - } - - controller.enqueue(Buffer.from(chunk)); - } - - if (finished) { - // Request finished — read any remaining bytes then close. - while (true) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offset); - if (bytesRead === 0) break; - const chunk = buf.subarray(0, bytesRead); - offset += bytesRead; - const text = chunk.toString("utf8"); - - if (text.includes(eofSymbol)) { - const beforeEof = text.substring(0, text.indexOf(eofSymbol)); - if (beforeEof) { - controller.enqueue(Buffer.from(beforeEof, "utf8")); - } - controller.close(); - return; - } - - controller.enqueue(Buffer.from(chunk)); - } - - if (upstreamError && !errored) { - errored = true; - controller.error(upstreamError); - return; - } - - controller.close(); - return; - } - - // No data yet and request still running — brief pause before retry. - await sleep(25); - } - } catch (err) { - if (!errored) { - errored = true; - controller.error(err instanceof Error ? err : new Error(String(err))); - } - } finally { - await fd.close().catch(() => {}); - await cleanupTempPath(path); - if (signal) signal.removeEventListener("abort", onAbort); - } - }, - }); -} - -function sleep(ms: number): Promise { - return new Promise((resolve) => setTimeout(resolve, ms)); -} +export const tlsClientModule = createTlsClientModule({ + providerName: "Grok", + tlsProfile: "chrome_146", + domain: "https://grok.com", + tempDirPrefix: "grok-stream-", + tailFileVariant: "B1", + responseValidation: "cf", + exportCloudflareCheck: true, + defaultTimeoutMs: DEFAULT_TIMEOUT_MS, + hardTimeoutGraceMs: HARD_TIMEOUT_GRACE_MS, +}); + +export const tlsFetchGrok = (url: string, options: TlsFetchOptions = {}): Promise => + tlsClientModule.tlsFetch(url, options); + +export const __setTlsFetchOverrideForTesting = tlsClientModule.__setTlsFetchOverrideForTesting; + +export { TlsClientHangError, TlsClientUnavailableError } from "./tlsClientBase.ts"; +export type { TlsFetchOptions, TlsFetchResult } from "./tlsClientBase.ts"; +export { isCloudflareChallenge } from "./tlsClientBase.ts"; diff --git a/open-sse/services/lmarenaTlsClient.ts b/open-sse/services/lmarenaTlsClient.ts index 496579606e0..131acb550e6 100644 --- a/open-sse/services/lmarenaTlsClient.ts +++ b/open-sse/services/lmarenaTlsClient.ts @@ -1,606 +1,43 @@ /** * Browser-TLS-impersonating HTTP client for arena.ai. * - * Why this exists: LMArena sits behind Cloudflare Enterprise which pins - * `cf_clearance` to the client's TLS fingerprint (JA3/JA4) + HTTP/2 SETTINGS - * frame ordering. Node's Undici fetch presents an obvious "not a browser" - * handshake and gets challenged with a 403 even with a valid arena session - * cookie (and often a browser-minted `cf_clearance`). This module wraps - * `tls-client-node` (bogdanfinn/tls-client) to send a Chrome handshake instead. - * - * Mirrors `grokTlsClient.ts` / `perplexityTlsClient.ts` as an independent - * module so changes here cannot regress those production paths. - * - * Note: Arena may still require a browser-issued reCAPTCHA v3 token on - * create-evaluation; TLS alone is necessary but not always sufficient. + * Thin re-export over the shared `tlsClientBase.ts` factory + * (`createTlsClientModule`). All provider-agnostic logic (sidecar lifecycle, + * streaming tail-file, proxy resolution, error classes, Cloudflare challenge + * detection) lives in the base module; this file supplies only LMArena-specific + * config and preserves the original public export surface. */ -import { tmpdir } from "node:os"; -import { join, dirname } from "node:path"; -import { mkdtemp, open, unlink, rmdir, stat } from "node:fs/promises"; -import { randomUUID } from "node:crypto"; -import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; +import { + createTlsClientModule, + type TlsFetchOptions, + type TlsFetchResult, +} from "./tlsClientBase.ts"; -let clientPromise: Promise | null = null; -let exitHookInstalled = false; - -// Newest Chrome JA3 profile shipped by tls-client-node (no chrome_147+ yet). -// HTTP User-Agent / Sec-Ch-Ua track Chrome 150 separately in models.ts. -const LMARENA_PROFILE = "chrome_146"; -// Fixed timeouts (same defaults as other TLS sidecars). No extra env knobs — -// env-doc-sync must not grow for provider-local constants. const DEFAULT_TIMEOUT_MS = 60_000; -// Grace period added to the binding's wire-level timeout before our JS-level -// hard timeout fires. Under healthy operation `tls-client-node` honors -// `timeoutMilliseconds` and rejects on its own; the JS-level race only wins -// when the koffi-loaded native library is wedged (which the binding's own -// timer can't escape). const HARD_TIMEOUT_GRACE_MS = 10_000; -function installExitHook(): void { - if (exitHookInstalled) return; - exitHookInstalled = true; - const stop = async () => { - if (clientPromise === null) return; - try { - const c = (await clientPromise) as { stop?: () => Promise }; - await c.stop?.(); - } catch { - // ignore - } - }; - process.once("beforeExit", stop); - process.once("SIGINT", () => { - void stop(); - }); - process.once("SIGTERM", () => { - void stop(); - }); -} - -/** - * Drop the cached client so the next `getClient()` call respawns it. Called - * when a request observes the native binding has wedged — releasing the - * reference lets a fresh TLSClient (and a fresh koffi load) take over without - * a process restart. - */ -function resetClientCache(): void { - clientPromise = null; -} - -export class TlsClientHangError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientHangError"; - } -} - -/** - * Race a `client.request()` promise against (a) a JS-level hard timeout and - * (b) the caller's abort signal. The native binding's `timeoutMilliseconds` - * already covers the wire path; this guards the case where the koffi binding - * itself deadlocks (observed after sustained load), where neither the - * binding's own timer nor a post-call `signal.aborted` re-check can recover. - */ -async function raceWithTimeout( - promise: Promise, - timeoutMs: number, - signal: AbortSignal | null | undefined -): Promise { - let timer: ReturnType | null = null; - let abortListener: (() => void) | null = null; - try { - const racers: Promise[] = [ - promise, - new Promise((_, reject) => { - timer = setTimeout(() => { - reject( - new TlsClientHangError( - `tls-client-node call exceeded ${timeoutMs}ms — native binding likely deadlocked` - ) - ); - }, timeoutMs); - }), - ]; - if (signal) { - racers.push( - new Promise((_, reject) => { - if (signal.aborted) { - reject(makeAbortError(signal)); - return; - } - abortListener = () => reject(makeAbortError(signal)); - signal.addEventListener("abort", abortListener, { once: true }); - }) - ); - } - return await Promise.race(racers); - } finally { - if (timer) clearTimeout(timer); - if (signal && abortListener) signal.removeEventListener("abort", abortListener); - } -} - -async function getClient(): Promise<{ - request: (url: string, opts: Record) => Promise; -}> { - if (!clientPromise) { - clientPromise = (async () => { - try { - const mod = await import("tls-client-node"); - const TLSClient = (mod as { TLSClient: new (opts?: Record) => unknown }) - .TLSClient; - // Native mode loads the shared library directly via koffi, avoiding the - // managed sidecar's localhost HTTP calls that OmniRoute's global fetch - // proxy patch interferes with. - const client = new TLSClient(buildNativeTlsClientOptions()) as { - start: () => Promise; - request: (url: string, opts: Record) => Promise; - }; - await client.start(); - - installExitHook(); - return client; - } catch (err) { - clientPromise = null; - const msg = err instanceof Error ? err.message : String(err); - throw new TlsClientUnavailableError( - `TLS impersonation client failed to start: ${msg}. ` + - `Verify tls-client-node is installed and its native binary downloaded.` - ); - } - })(); - } - return clientPromise as Promise<{ - request: (url: string, opts: Record) => Promise; - }>; -} - -interface TlsResponseLike { - status: number; - headers: Record; - body: string; // for non-streaming requests, the full response body - cookies?: Record; - text: () => Promise; - bytes: () => Promise; - json: () => Promise; -} - -export class TlsClientUnavailableError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientUnavailableError"; - } -} - -export interface TlsFetchOptions { - method?: "GET" | "POST" | "PUT" | "PATCH" | "DELETE"; - headers?: Record; - body?: string; - timeoutMs?: number; - signal?: AbortSignal | null; - /** - * If true, the response body is streamed to a temp file and exposed as a - * ReadableStream. Use for NDJSON streaming responses (the - * LMArena conversation endpoint). Otherwise, the full body is read into memory. - */ - stream?: boolean; - /** EOF marker the upstream sends to signal end of stream (default: "[DONE]"). */ - streamEofSymbol?: string; - /** - * Optional upstream proxy URL (`http://user:pass@host:port` or - * `socks5://...`). When set, the request is tunneled through this proxy - * before reaching arena.ai. - * - * Resolution order: - * 1. `options.proxyUrl` (per-call override from caller) - * 2. `process.env.OMNIROUTE_TLS_PROXY_URL` (single-flag opt-in) - * 3. `process.env.HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (POSIX-standard fallback) - * - * The native `tls-client-node` binding does **not** consult Go's - * `http.ProxyFromEnvironment`, so the env vars need to be plumbed in here at - * the JS layer. - */ - proxyUrl?: string; -} - -import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; -import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; - -/** - * Resolve the proxy URL for a tls-client request. Per-call value wins; - * otherwise we use the standard proxy fetch resolution which reads from - * the dashboard AsyncLocalStorage context or falls back to env vars. - * - * Fail-closed: if resolution throws (e.g. a configured socks5 proxy with - * ENABLE_SOCKS5_PROXY=false), this rethrows rather than returning undefined — - * undefined would let the native binding connect directly and leak the real IP. - */ -function resolveProxyUrl(perCall: string | undefined): string | undefined { - return resolveTlsClientProxyUrl("https://arena.ai", perCall, resolveProxyForRequest); -} - -export interface TlsFetchResult { - status: number; - headers: Headers; - /** Full response body as text — only populated for non-streaming requests. */ - text: string | null; - /** Streaming body — only populated when options.stream === true. */ - body: ReadableStream | null; -} - -// Test-only injection point. Tests call __setTlsFetchOverrideForTesting() -// to replace the real TLS client with a mock; production never touches this. -let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = - null; - -export function __setTlsFetchOverrideForTesting(fn: typeof testOverride): void { - testOverride = fn; -} - -function throwIfAborted(signal: AbortSignal | null | undefined): void { - if (signal?.aborted) throw makeAbortError(signal); -} - -function buildTlsRequestOptions(options: TlsFetchOptions): Record { - return { - method: options.method || "GET", - headers: options.headers || {}, - body: options.body, - tlsClientIdentifier: LMARENA_PROFILE, - timeoutMilliseconds: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, - followRedirects: true, - withRandomTLSExtensionOrder: true, - // Plumb proxy via options — tls-client-node does not read HTTP_PROXY env. - proxyUrl: resolveProxyUrl(options.proxyUrl), - }; -} - -function hardTimeoutMs(options: TlsFetchOptions): number { - return (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS; -} - -async function tlsFetchNonStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - options: TlsFetchOptions -): Promise { - let tlsResponse: TlsResponseLike; - try { - tlsResponse = await raceWithTimeout( - client.request(url, requestOptions), - hardTimeoutMs(options), - options.signal ?? null - ); - } catch (err) { - if (err instanceof TlsClientHangError) resetClientCache(); - throw err; - } - throwIfAborted(options.signal); - return { - status: tlsResponse.status, - headers: toHeaders(tlsResponse.headers), - text: tlsResponse.body, - body: null, - }; -} - -/** - * Make a single HTTP request to arena.ai with a Chrome-like TLS fingerprint. - * Throws TlsClientUnavailableError if the native binary failed to load. - */ -export async function tlsFetchLMArena( +export const tlsClientModule = createTlsClientModule({ + providerName: "LMArena", + tlsProfile: "chrome_146", + domain: "https://lmarena.ai", + // LMArena's proxy resolution domain is hardcoded to arena.ai, not the config domain. + proxyDomainOverride: "https://arena.ai", + tempDirPrefix: "LMArena-stream-", + tailFileVariant: "B2", + responseValidation: "cf", + exportCloudflareCheck: true, + defaultTimeoutMs: DEFAULT_TIMEOUT_MS, + hardTimeoutGraceMs: HARD_TIMEOUT_GRACE_MS, +}); + +export const tlsFetchLMArena = ( url: string, options: TlsFetchOptions = {} -): Promise { - if (testOverride) return testOverride(url, options); - throwIfAborted(options.signal); - const client = await getClient(); - throwIfAborted(options.signal); - - const requestOptions = buildTlsRequestOptions(options); - if (options.stream) { - return tlsFetchStreaming( - client, - url, - requestOptions, - options.streamEofSymbol, - options.signal ?? null, - hardTimeoutMs(options) - ); - } - return tlsFetchNonStreaming(client, url, requestOptions, options); -} - -function makeAbortError(signal: AbortSignal): Error { - const reason = signal.reason; - if (reason instanceof Error) return reason; - const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); - err.name = "AbortError"; - return err; -} - -function toHeaders(raw: Record): Headers { - const h = new Headers(); - for (const [k, vs] of Object.entries(raw || {})) { - for (const v of vs) h.append(k, v); - } - return h; -} - -/** - * Returns true if the response body is a Cloudflare challenge/interstitial page - * rather than a real LMArena response. From VPS/datacenter IPs a valid cookie - * still gets a 403 "Request rejected by anti-bot rules." JSON; distinguishing - * it from a genuine auth failure lets the caller surface an actionable error - * (issue #3180). - * - * Exported so the executor and the connection validator share one detector. - */ -export function isCloudflareChallenge(text: string | null | undefined): boolean { - if (!text) return false; - return /just a moment|window\._cf_chl_opt|challenges\.cloudflare\.com|attention required|cf-chl/i.test( - text - ); -} - -// ─── Streaming via temp file ──────────────────────────────────────────────── -// tls-client-node's streaming primitive writes the response body chunk-by-chunk -// to a file path, terminating when the upstream sends `streamOutputEOFSymbol`. -// We tail the file from a worker and surface the bytes as a ReadableStream. - -async function tlsFetchStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS -): Promise { - const dir = await mkdtemp(join(tmpdir(), "LMArena-stream-")); - const path = join(dir, `${randomUUID()}.ndjson`); - - const streamOpts = { - ...requestOptions, - streamOutputPath: path, - streamOutputBlockSize: 1024, - streamOutputEOFSymbol: eofSymbol, - }; - - // Kick off the request without awaiting — tls-client writes the body to - // `path` chunk-by-chunk while the call runs. The Promise resolves when the - // request fully completes (full body written). Wrapping in raceWithTimeout - // guarantees this promise eventually settles even if the koffi binding - // wedges; on hang we reset the singleton so the next request respawns. - let resetOnHang = true; - const requestPromise = raceWithTimeout( - client.request(url, streamOpts), - hardTimeoutMs, - signal - ).catch((err: unknown) => { - if (resetOnHang && err instanceof TlsClientHangError) { - resetClientCache(); - resetOnHang = false; - } - // Re-throw so downstream consumers (waitForContent, tailFile) observe - // the rejection and surface it instead of treating the stream as having - // ended cleanly. - throw err; - }); - - // Wait for the file to exist AND have at least one byte. - const ready = await waitForContent(path, 5_000, requestPromise); - if (!ready) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Peek at the first bytes to distinguish a genuine NDJSON stream from a - // Cloudflare challenge page or an HTML error response that tls-client-node - // streamed to the temp file with a 200 status. - const peek = await readFirstBytes(path, 256); - if (isCloudflareChallenge(peek)) { - await cleanupTempPath(path); - return { - status: 403, - headers: new Headers({ "Content-Type": "text/html" }), - text: peek, - body: null, - }; - } - if (peek.trimStart().startsWith("<")) { - // HTML error page (not a challenge) — surface as a non-2xx so the executor - // can emit a proper SSE error chunk instead of feeding HTML to the NDJSON - // parser. - await cleanupTempPath(path); - return { - status: 502, - headers: new Headers({ "Content-Type": "text/html" }), - text: peek, - body: null, - }; - } - - // Looks like NDJSON — start tailing. The requestPromise will eventually - // resolve with the real upstream status; tailFile propagates non-2xx errors - // into the stream so the consumer sees them instead of a truncated success. - const stream = tailFile(path, eofSymbol, requestPromise, signal); - const headers = new Headers({ - "Content-Type": "application/x-ndjson", - "Cache-Control": "no-cache", - }); - return { status: 200, headers, text: null, body: stream }; -} - -async function cleanupTempPath(path: string): Promise { - await unlink(path).catch(() => {}); - await rmdir(dirname(path)).catch(() => {}); -} - -async function readFirstBytes(path: string, n: number): Promise { - const fd = await open(path, "r"); - try { - const buf = Buffer.alloc(n); - const { bytesRead } = await fd.read(buf, 0, n, 0); - return buf.subarray(0, bytesRead).toString("utf8"); - } finally { - await fd.close().catch(() => {}); - } -} - -/** - * Wait for the streaming output file to exist AND contain at least one byte. - * Returns false if the request settles before any bytes arrive (so the caller - * can drain `requestPromise` and surface the real upstream status). Returns - * true as soon as the file has data — even one byte is enough for the NDJSON - * heuristic to give a useful answer. - */ -async function waitForContent( - path: string, - timeoutMs: number, - requestPromise: Promise -): Promise { - let requestSettled = false; - requestPromise.then( - () => { - requestSettled = true; - }, - () => { - requestSettled = true; - } - ); - const start = Date.now(); - while (Date.now() - start < timeoutMs) { - try { - const s = await stat(path); - if (s.size > 0) return true; - } catch { - // file doesn't exist yet - } - // If the request finished without producing any bytes, no point waiting - // out the rest of the timeout — let the caller drain it. - if (requestSettled) return false; - await sleep(25); - } - return false; -} - -/** Enqueue chunk bytes, splitting off an EOF symbol when present. Returns true if closed. */ -function enqueueChunkMaybeEof( - controller: ReadableStreamDefaultController, - chunk: Buffer, - eofSymbol: string -): boolean { - const text = chunk.toString("utf8"); - if (!text.includes(eofSymbol)) { - controller.enqueue(Buffer.from(chunk)); - return false; - } - const beforeEof = text.substring(0, text.indexOf(eofSymbol)); - if (beforeEof) controller.enqueue(Buffer.from(beforeEof, "utf8")); - controller.close(); - return true; -} - -type FileHandle = Awaited>; - -async function drainRemaining( - fd: FileHandle, - buf: Buffer, - offsetRef: { offset: number }, - controller: ReadableStreamDefaultController, - eofSymbol: string -): Promise<"closed" | "drained"> { - while (true) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offsetRef.offset); - if (bytesRead === 0) return "drained"; - const chunk = buf.subarray(0, bytesRead); - offsetRef.offset += bytesRead; - if (enqueueChunkMaybeEof(controller, chunk, eofSymbol)) return "closed"; - } -} - -function tailFile( - path: string, - eofSymbol: string, - done: Promise, - signal: AbortSignal | null = null -): ReadableStream { - return new ReadableStream({ - async start(controller) { - const fd = await open(path, "r"); - const buf = Buffer.alloc(64 * 1024); - const offsetRef = { offset: 0 }; - let finished = false; - let aborted = false; - let upstreamError: Error | null = null; - let errored = false; - - done.then( - () => { - finished = true; - }, - (err) => { - upstreamError = err instanceof Error ? err : new Error(String(err)); - finished = true; - } - ); - - const onAbort = () => { - aborted = true; - }; - if (signal) { - if (signal.aborted) aborted = true; - else signal.addEventListener("abort", onAbort, { once: true }); - } - - try { - while (!aborted) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offsetRef.offset); - if (bytesRead > 0) { - const chunk = buf.subarray(0, bytesRead); - offsetRef.offset += bytesRead; - if (enqueueChunkMaybeEof(controller, chunk, eofSymbol)) return; - } - - if (!finished) { - await sleep(25); - continue; - } +): Promise => tlsClientModule.tlsFetch(url, options); - const drained = await drainRemaining(fd, buf, offsetRef, controller, eofSymbol); - if (drained === "closed") return; - if (upstreamError && !errored) { - errored = true; - controller.error(upstreamError); - return; - } - controller.close(); - return; - } - } catch (err) { - if (!errored) { - errored = true; - controller.error(err instanceof Error ? err : new Error(String(err))); - } - } finally { - await fd.close().catch(() => {}); - await cleanupTempPath(path); - if (signal) signal.removeEventListener("abort", onAbort); - } - }, - }); -} +export const __setTlsFetchOverrideForTesting = tlsClientModule.__setTlsFetchOverrideForTesting; -function sleep(ms: number): Promise { - return new Promise((resolve) => setTimeout(resolve, ms)); -} +export { TlsClientHangError, TlsClientUnavailableError } from "./tlsClientBase.ts"; +export type { TlsFetchOptions, TlsFetchResult } from "./tlsClientBase.ts"; +export { isCloudflareChallenge } from "./tlsClientBase.ts"; diff --git a/open-sse/services/model.ts b/open-sse/services/model.ts index 9df7d14aaf4..04be2c9f352 100644 --- a/open-sse/services/model.ts +++ b/open-sse/services/model.ts @@ -818,11 +818,15 @@ async function resolveModelByProviderInference(modelId: string, extendedContext: return { provider: "claude", model: modelId, extendedContext }; } // Claude models → Anthropic provider (canonical source for Claude models) - return { provider: "anthropic", model: modelId, extendedContext }; + if (activeProviders?.has("anthropic")) { + return { provider: "anthropic", model: modelId, extendedContext }; + } } if (/^gemini-/i.test(modelId) || /^gemma-/i.test(modelId)) { // Gemini/Gemma models → Gemini provider - return { provider: "gemini", model: modelId, extendedContext }; + if (activeProviders?.has("gemini")) { + return { provider: "gemini", model: modelId, extendedContext }; + } } // Last resort: no provider could be inferred — return a clear error instead diff --git a/open-sse/services/notionTlsClient.ts b/open-sse/services/notionTlsClient.ts index a11a676b1f9..2dc56e5f358 100644 --- a/open-sse/services/notionTlsClient.ts +++ b/open-sse/services/notionTlsClient.ts @@ -1,594 +1,43 @@ /** * Browser-TLS-impersonating HTTP client for app.notion.com. * - * Why this exists: Notion AI sits behind the same Cloudflare Enterprise - * configuration as ChatGPT — it pins access to the client's TLS fingerprint - * (JA3/JA4) + HTTP/2 SETTINGS frame ordering. Node's Undici fetch presents an - * obvious "not a browser" handshake and gets challenged with a 403 "Just a - * moment..." page from VPS/datacenter IPs — even with a valid session cookie. - * This module wraps `tls-client-node` (native shared library built from - * bogdanfinn/tls-client) to send a Firefox handshake instead. (issue #2459) - * - * Mirrors `claudeTlsClient.ts` / `perplexityTlsClient.ts`; kept as an independent module so changes here - * cannot regress the production chatgpt-web path. The first call lazily starts - * the managed sidecar; subsequent calls reuse a singleton TLSClient. Process - * exit hooks stop the sidecar cleanly. + * Thin re-export over the shared `tlsClientBase.ts` factory + * (`createTlsClientModule`). All provider-agnostic logic (sidecar lifecycle, + * streaming tail-file, proxy resolution, error classes, SSE detection, + * Cloudflare challenge detection) lives in the base module; this file supplies + * only Notion-specific config and preserves the original public export surface. */ -import { tmpdir } from "node:os"; -import { join } from "node:path"; -import { mkdtemp, open, unlink, rmdir, stat } from "node:fs/promises"; -import { randomUUID } from "node:crypto"; -import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; - -let clientPromise: Promise | null = null; -let exitHookInstalled = false; +import { + createTlsClientModule, + type TlsFetchOptions, + type TlsFetchResult, +} from "./tlsClientBase.ts"; -const NOTION_PROFILE = "chrome_146"; // matches the Chrome UA we send const DEFAULT_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_NOTION_TLS_TIMEOUT_MS || "", 10) || 30_000; -// Grace period added to the binding's wire-level timeout before our JS-level -// hard timeout fires. Under healthy operation `tls-client-node` honors -// `timeoutMilliseconds` and rejects on its own; the JS-level race only wins -// when the koffi-loaded native library is wedged (which the binding's own -// timer can't escape). Keep the grace small so users don't wait noticeably -// longer than the configured timeout when the binding is dead. const HARD_TIMEOUT_GRACE_MS = Number.parseInt(process.env.OMNIROUTE_NOTION_TLS_GRACE_MS || "", 10) || 10_000; -function installExitHook(): void { - if (exitHookInstalled) return; - exitHookInstalled = true; - const stop = async () => { - if (!clientPromise) return; - try { - const c = (await clientPromise) as { stop?: () => Promise }; - await c.stop?.(); - } catch { - // ignore - } - }; - process.once("beforeExit", stop); - process.once("SIGINT", () => { - void stop(); - }); - process.once("SIGTERM", () => { - void stop(); - }); -} - -/** - * Drop the cached client so the next `getClient()` call respawns it. Called - * when a request observes the native binding has wedged — releasing the - * reference lets a fresh TLSClient (and a fresh koffi load) take over without - * a process restart. - */ -function resetClientCache(): void { - clientPromise = null; -} - -export class TlsClientHangError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientHangError"; - } -} - -/** - * Race a `client.request()` promise against (a) a JS-level hard timeout and - * (b) the caller's abort signal. The native binding's `timeoutMilliseconds` - * already covers the wire path; this guards the case where the koffi binding - * itself deadlocks (observed after sustained load), where neither the - * binding's own timer nor a post-call `signal.aborted` re-check can recover. - */ -async function raceWithTimeout( - promise: Promise, - timeoutMs: number, - signal: AbortSignal | null | undefined -): Promise { - let timer: ReturnType | null = null; - let abortListener: (() => void) | null = null; - try { - const racers: Promise[] = [ - promise, - new Promise((_, reject) => { - timer = setTimeout(() => { - reject( - new TlsClientHangError( - `tls-client-node call exceeded ${timeoutMs}ms — native binding likely deadlocked` - ) - ); - }, timeoutMs); - }), - ]; - if (signal) { - racers.push( - new Promise((_, reject) => { - if (signal.aborted) { - reject(makeAbortError(signal)); - return; - } - abortListener = () => reject(makeAbortError(signal)); - signal.addEventListener("abort", abortListener, { once: true }); - }) - ); - } - return await Promise.race(racers); - } finally { - if (timer) clearTimeout(timer); - if (signal && abortListener) signal.removeEventListener("abort", abortListener); - } -} - -async function getClient(): Promise<{ - request: (url: string, opts: Record) => Promise; -}> { - if (!clientPromise) { - clientPromise = (async () => { - try { - const mod = await import("tls-client-node"); - const TLSClient = (mod as { TLSClient: new (opts?: Record) => unknown }) - .TLSClient; - // Native mode loads the shared library directly via koffi, avoiding the - // managed sidecar's localhost HTTP calls that OmniRoute's global fetch - // proxy patch interferes with. - const client = new TLSClient(buildNativeTlsClientOptions()) as { - start: () => Promise; - request: (url: string, opts: Record) => Promise; - }; - await client.start(); - - installExitHook(); - return client; - } catch (err) { - clientPromise = null; - const msg = err instanceof Error ? err.message : String(err); - throw new TlsClientUnavailableError( - `TLS impersonation client failed to start: ${msg}. ` + - `Verify tls-client-node is installed and its native binary downloaded.` - ); - } - })(); - } - return clientPromise as Promise<{ - request: (url: string, opts: Record) => Promise; - }>; -} - -interface TlsResponseLike { - status: number; - headers: Record; - body: string; // for non-streaming requests, the full response body - cookies?: Record; - text: () => Promise; - bytes: () => Promise; - json: () => Promise; -} - -export class TlsClientUnavailableError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientUnavailableError"; - } -} - -export interface TlsFetchOptions { - method?: "GET" | "POST" | "PUT" | "PATCH" | "DELETE"; - headers?: Record; - body?: string; - timeoutMs?: number; - signal?: AbortSignal | null; - /** - * If true, the response body is streamed to a temp file and exposed as a - * ReadableStream. Use for SSE responses (the runInferenceTranscript - * endpoint). Otherwise, the full body is read into memory. - */ - stream?: boolean; - /** EOF marker the upstream sends to signal end of stream (default: "[DONE]"). */ - streamEofSymbol?: string; - /** - * Optional upstream proxy URL (`http://user:pass@host:port` or - * `socks5://...`). When set, the request is tunneled through this proxy - * before reaching notion.so. - * - * Resolution order: - * 1. `options.proxyUrl` (per-call override from caller) - * 2. `process.env.OMNIROUTE_TLS_PROXY_URL` (single-flag opt-in) - * 3. `process.env.HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (POSIX-standard fallback) - * - * The native `tls-client-node` binding does **not** consult Go's - * `http.ProxyFromEnvironment`, so the env vars need to be plumbed in here at - * the JS layer. - */ - proxyUrl?: string; -} - -import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; -import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; - -/** - * Resolve the proxy URL for a tls-client request. Per-call value wins; - * otherwise we use the standard proxy fetch resolution which reads from - * the dashboard AsyncLocalStorage context or falls back to env vars. - * - * Fail-closed: if resolution throws (e.g. a configured socks5 proxy with - * ENABLE_SOCKS5_PROXY=false), this rethrows rather than returning undefined — - * undefined would let the native binding connect directly and leak the real IP. - */ -function resolveProxyUrl(perCall: string | undefined): string | undefined { - return resolveTlsClientProxyUrl("https://app.notion.com", perCall, resolveProxyForRequest); -} - -export interface TlsFetchResult { - status: number; - headers: Headers; - /** Full response body as text — only populated for non-streaming requests. */ - text: string | null; - /** Streaming body — only populated when options.stream === true. */ - body: ReadableStream | null; -} - -// Test-only injection point. Tests call __setTlsFetchOverrideForTesting() -// to replace the real TLS client with a mock; production never touches this. -let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = - null; - -export function __setTlsFetchOverrideForTesting(fn: typeof testOverride): void { - testOverride = fn; -} - -/** - * Make a single HTTP request to notion.so with a Chrome-like TLS fingerprint. - * - * Throws TlsClientUnavailableError if the native binary failed to load. - */ -export async function tlsFetchNotion( +export const tlsClientModule = createTlsClientModule({ + providerName: "Notion", + tlsProfile: "chrome_146", + domain: "https://app.notion.com", + tempDirPrefix: "pplx-stream-", + tailFileVariant: "A", + responseValidation: "sse", + exportCloudflareCheck: true, + defaultTimeoutMs: DEFAULT_TIMEOUT_MS, + hardTimeoutGraceMs: HARD_TIMEOUT_GRACE_MS, +}); + +export const tlsFetchNotion = ( url: string, options: TlsFetchOptions = {} -): Promise { - if (testOverride) return testOverride(url, options); - // Honor abort signals up-front. tls-client-node's koffi binding doesn't - // accept an AbortSignal mid-flight (the binary call is opaque), so the best - // we can do is bail before issuing the call. We also re-check after — if - // the caller aborted while the upstream was running, throw rather than - // returning a stale response so the caller doesn't try to use it. - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - const client = await getClient(); - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - - const requestOptions: Record = { - method: options.method || "GET", - headers: options.headers || {}, - body: options.body, - tlsClientIdentifier: NOTION_PROFILE, - timeoutMilliseconds: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, - followRedirects: true, - withRandomTLSExtensionOrder: true, - // Plumb the configured proxy through to the native binding. tls-client-node - // consults `proxyUrl` in the per-call options (it does NOT auto-pick up - // HTTP_PROXY / HTTPS_PROXY env), so callers / env have to be threaded in - // explicitly. See `resolveProxyUrl()` for the lookup order. - proxyUrl: resolveProxyUrl(options.proxyUrl), - }; - - if (options.stream) { - return await tlsFetchStreaming( - client, - url, - requestOptions, - options.streamEofSymbol, - options.signal ?? null, - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS - ); - } - - let tlsResponse: TlsResponseLike; - try { - tlsResponse = await raceWithTimeout( - client.request(url, requestOptions), - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS, - options.signal ?? null - ); - } catch (err) { - if (err instanceof TlsClientHangError) { - // The native binding is wedged — drop the singleton so the next - // request respawns a fresh client (and a fresh koffi load). - resetClientCache(); - } - throw err; - } - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - return { - status: tlsResponse.status, - headers: toHeaders(tlsResponse.headers), - text: tlsResponse.body, - body: null, - }; -} - -function makeAbortError(signal: AbortSignal): Error { - const reason = signal.reason; - if (reason instanceof Error) return reason; - const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); - err.name = "AbortError"; - return err; -} - -function toHeaders(raw: Record): Headers { - const h = new Headers(); - for (const [k, vs] of Object.entries(raw || {})) { - for (const v of vs) h.append(k, v); - } - return h; -} - -/** - * Returns true if the response body is a Cloudflare challenge/interstitial page - * rather than a real Perplexity response. From VPS/datacenter IPs a valid cookie - * still gets a 403 "Just a moment..." HTML page; distinguishing it from a genuine - * auth failure lets the caller surface an actionable error (issue #2459). - * - * Exported so the executor and the connection validator share one detector. - */ -export function isCloudflareChallenge(text: string | null | undefined): boolean { - if (!text) return false; - return /just a moment|window\._cf_chl_opt|challenges\.cloudflare\.com|attention required|cf-chl/i.test( - text - ); -} - -// ─── Streaming via temp file ──────────────────────────────────────────────── -// tls-client-node's streaming primitive writes the response body chunk-by-chunk -// to a file path, terminating when the upstream sends `streamOutputEOFSymbol`. -// We tail the file from a worker and surface the bytes as a ReadableStream. - -async function tlsFetchStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS -): Promise { - const dir = await mkdtemp(join(tmpdir(), "pplx-stream-")); - const path = join(dir, `${randomUUID()}.sse`); - - const streamOpts = { - ...requestOptions, - streamOutputPath: path, - streamOutputBlockSize: 1024, - streamOutputEOFSymbol: eofSymbol, - }; - - // Kick off the request without awaiting — tls-client writes the body to - // `path` chunk-by-chunk while the call runs. The Promise resolves when the - // request fully completes (full body written). Wrapping in raceWithTimeout - // guarantees this promise eventually settles even if the koffi binding - // wedges; on hang we reset the singleton so the next request respawns. - let resetOnHang = true; - const requestPromise = raceWithTimeout( - client.request(url, streamOpts), - hardTimeoutMs, - signal - ).catch((err: unknown) => { - if (resetOnHang && err instanceof TlsClientHangError) { - resetClientCache(); - resetOnHang = false; - } - // Re-throw so downstream consumers (waitForContent, tailFile) observe - // the rejection and surface it instead of treating the stream as having - // ended cleanly. - throw err; - }); - - // Wait for the file to exist AND have at least one byte. tls-client-node - // creates the output file when the request starts, but the file can be - // empty for a brief window before the first body chunk lands — peeking - // during that window would return "" and misclassify the response as - // non-SSE, dropping us into the buffered-wait branch and silently turning - // a streaming request into a buffered one. Waiting for content avoids - // that race; if the request actually fails before producing any bytes, - // the timeout falls through to the requestPromise drain below (returning - // the real upstream status). - const ready = await waitForContent(path, 5_000, requestPromise); - if (!ready) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Peek the first bytes to decide whether this looks like SSE. Anything - // that doesn't positively look like SSE (JSON `{...}`, HTML `<...>`, plain - // text rate-limit messages, Cloudflare challenge pages, etc.) gets surfaced - // as a non-streaming response so the executor sees the real upstream status - // and body — otherwise non-2xx error pages get silently treated as 200 OK - // and the SSE parser produces an empty completion. - const peek = await readFirstBytes(path, 256); - if (!looksLikeSse(peek)) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Looks like SSE — start tailing. SSE bodies in practice are always 2xx; - // tls-client-node doesn't expose response status separately from full-body - // completion, so we report 200 and let the SSE parser consume the stream. - const stream = tailFile(path, eofSymbol, requestPromise, signal); - const headers = new Headers({ - "Content-Type": "text/event-stream", - "Cache-Control": "no-cache", - }); - return { status: 200, headers, text: null, body: stream }; -} - -/** - * Returns true if the peeked response body looks like an SSE stream — i.e., - * begins (after any leading whitespace) with one of the SSE field markers - * (`data:`, `event:`, `id:`, `retry:`) or a comment line (`:`). - * - * Exported for tests. - */ -export function looksLikeSse(text: string): boolean { - const trimmed = text.replace(/^[\s\r\n]+/, ""); - if (!trimmed) return false; - if (trimmed.startsWith(":")) return true; - return /^(data|event|id|retry):/i.test(trimmed); -} - -async function cleanupTempPath(path: string): Promise { - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); -} - -async function readFirstBytes(path: string, n: number): Promise { - const fd = await open(path, "r"); - try { - const buf = Buffer.alloc(n); - const { bytesRead } = await fd.read(buf, 0, n, 0); - return buf.subarray(0, bytesRead).toString("utf8"); - } finally { - await fd.close().catch(() => {}); - } -} - -/** - * Wait for the streaming output file to exist AND contain at least one byte. - * Returns false if the request settles before any bytes arrive (so the caller - * can drain `requestPromise` and surface the real upstream status). Returns - * true as soon as the file has data — even one byte is enough for the SSE - * heuristic to give a useful answer. - */ -async function waitForContent( - path: string, - timeoutMs: number, - requestPromise: Promise -): Promise { - let requestSettled = false; - requestPromise.then( - () => { - requestSettled = true; - }, - () => { - requestSettled = true; - } - ); - const start = Date.now(); - while (Date.now() - start < timeoutMs) { - try { - const s = await stat(path); - if (s.size > 0) return true; - } catch { - // file doesn't exist yet - } - // If the request finished without producing any bytes, no point waiting - // out the rest of the timeout — let the caller drain it. - if (requestSettled) return false; - await sleep(25); - } - return false; -} - -function tailFile( - path: string, - eofSymbol: string, - done: Promise, - signal: AbortSignal | null = null -): ReadableStream { - return new ReadableStream({ - async start(controller) { - const fd = await open(path, "r"); - const buf = Buffer.alloc(64 * 1024); - let offset = 0; - let finished = false; - let aborted = false; - let upstreamError: Error | null = null; - - // Track request settlement, capturing both fulfillment and rejection. - // Without the rejection branch, a mid-stream tls-client-node error - // becomes an unhandledRejection — the stream cleans up silently and - // the consumer sees what looks like a successful truncated response. - done.then( - () => { - finished = true; - }, - (err) => { - upstreamError = err instanceof Error ? err : new Error(String(err)); - finished = true; - } - ); - - // If the caller aborts, stop tailing immediately. - const onAbort = () => { - aborted = true; - }; - if (signal) { - if (signal.aborted) aborted = true; - else signal.addEventListener("abort", onAbort, { once: true }); - } +): Promise => tlsClientModule.tlsFetch(url, options); - let errored = false; - try { - while (!aborted) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offset); - if (bytesRead > 0) { - const chunk = buf.subarray(0, bytesRead); - offset += bytesRead; - const text = chunk.toString("utf8"); - if (text.includes(eofSymbol)) { - const cutAt = text.indexOf(eofSymbol) + eofSymbol.length; - controller.enqueue(new Uint8Array(chunk.subarray(0, cutAt))); - break; - } - controller.enqueue(new Uint8Array(chunk)); - } else if (finished) { - // No more data and request completed. If the request rejected, - // surface the error so the consumer doesn't think the stream - // ended cleanly. - if (upstreamError) { - controller.error(upstreamError); - errored = true; - } - break; - } else { - await sleep(25); - } - } - } catch (err) { - controller.error(err); - errored = true; - } finally { - if (signal) signal.removeEventListener("abort", onAbort); - await fd.close().catch(() => {}); - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); - if (!errored) controller.close(); - } - }, - }); -} +export const __setTlsFetchOverrideForTesting = tlsClientModule.__setTlsFetchOverrideForTesting; -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); -} +export { TlsClientHangError, TlsClientUnavailableError } from "./tlsClientBase.ts"; +export type { TlsFetchOptions, TlsFetchResult } from "./tlsClientBase.ts"; +export { looksLikeSse, isCloudflareChallenge } from "./tlsClientBase.ts"; diff --git a/open-sse/services/perplexityTlsClient.ts b/open-sse/services/perplexityTlsClient.ts index 081ccb090ae..bc736476c33 100644 --- a/open-sse/services/perplexityTlsClient.ts +++ b/open-sse/services/perplexityTlsClient.ts @@ -1,594 +1,44 @@ /** * Browser-TLS-impersonating HTTP client for www.perplexity.ai. * - * Why this exists: Perplexity sits behind the same Cloudflare Enterprise - * configuration as ChatGPT — it pins access to the client's TLS fingerprint - * (JA3/JA4) + HTTP/2 SETTINGS frame ordering. Node's Undici fetch presents an - * obvious "not a browser" handshake and gets challenged with a 403 "Just a - * moment..." page from VPS/datacenter IPs — even with a valid session cookie. - * This module wraps `tls-client-node` (native shared library built from - * bogdanfinn/tls-client) to send a Firefox handshake instead. (issue #2459) - * - * Mirrors `chatgptTlsClient.ts`; kept as an independent module so changes here - * cannot regress the production chatgpt-web path. The first call lazily starts - * the managed sidecar; subsequent calls reuse a singleton TLSClient. Process - * exit hooks stop the sidecar cleanly. + * Thin re-export over the shared `tlsClientBase.ts` factory + * (`createTlsClientModule`). All provider-agnostic logic (sidecar lifecycle, + * streaming tail-file, proxy resolution, error classes, SSE detection, + * Cloudflare challenge detection) lives in the base module; this file supplies + * only Perplexity-specific config and preserves the original public export + * surface. */ -import { tmpdir } from "node:os"; -import { join } from "node:path"; -import { mkdtemp, open, unlink, rmdir, stat } from "node:fs/promises"; -import { randomUUID } from "node:crypto"; -import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; - -let clientPromise: Promise | null = null; -let exitHookInstalled = false; +import { + createTlsClientModule, + type TlsFetchOptions, + type TlsFetchResult, +} from "./tlsClientBase.ts"; -const PPLX_PROFILE = "firefox_148"; // matches the Firefox 148 UA we send const DEFAULT_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_PPLX_TLS_TIMEOUT_MS || "", 10) || 30_000; -// Grace period added to the binding's wire-level timeout before our JS-level -// hard timeout fires. Under healthy operation `tls-client-node` honors -// `timeoutMilliseconds` and rejects on its own; the JS-level race only wins -// when the koffi-loaded native library is wedged (which the binding's own -// timer can't escape). Keep the grace small so users don't wait noticeably -// longer than the configured timeout when the binding is dead. const HARD_TIMEOUT_GRACE_MS = Number.parseInt(process.env.OMNIROUTE_PPLX_TLS_GRACE_MS || "", 10) || 10_000; -function installExitHook(): void { - if (exitHookInstalled) return; - exitHookInstalled = true; - const stop = async () => { - if (!clientPromise) return; - try { - const c = (await clientPromise) as { stop?: () => Promise }; - await c.stop?.(); - } catch { - // ignore - } - }; - process.once("beforeExit", stop); - process.once("SIGINT", () => { - void stop(); - }); - process.once("SIGTERM", () => { - void stop(); - }); -} - -/** - * Drop the cached client so the next `getClient()` call respawns it. Called - * when a request observes the native binding has wedged — releasing the - * reference lets a fresh TLSClient (and a fresh koffi load) take over without - * a process restart. - */ -function resetClientCache(): void { - clientPromise = null; -} - -export class TlsClientHangError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientHangError"; - } -} - -/** - * Race a `client.request()` promise against (a) a JS-level hard timeout and - * (b) the caller's abort signal. The native binding's `timeoutMilliseconds` - * already covers the wire path; this guards the case where the koffi binding - * itself deadlocks (observed after sustained load), where neither the - * binding's own timer nor a post-call `signal.aborted` re-check can recover. - */ -async function raceWithTimeout( - promise: Promise, - timeoutMs: number, - signal: AbortSignal | null | undefined -): Promise { - let timer: ReturnType | null = null; - let abortListener: (() => void) | null = null; - try { - const racers: Promise[] = [ - promise, - new Promise((_, reject) => { - timer = setTimeout(() => { - reject( - new TlsClientHangError( - `tls-client-node call exceeded ${timeoutMs}ms — native binding likely deadlocked` - ) - ); - }, timeoutMs); - }), - ]; - if (signal) { - racers.push( - new Promise((_, reject) => { - if (signal.aborted) { - reject(makeAbortError(signal)); - return; - } - abortListener = () => reject(makeAbortError(signal)); - signal.addEventListener("abort", abortListener, { once: true }); - }) - ); - } - return await Promise.race(racers); - } finally { - if (timer) clearTimeout(timer); - if (signal && abortListener) signal.removeEventListener("abort", abortListener); - } -} - -async function getClient(): Promise<{ - request: (url: string, opts: Record) => Promise; -}> { - if (!clientPromise) { - clientPromise = (async () => { - try { - const mod = await import("tls-client-node"); - const TLSClient = (mod as { TLSClient: new (opts?: Record) => unknown }) - .TLSClient; - // Native mode loads the shared library directly via koffi, avoiding the - // managed sidecar's localhost HTTP calls that OmniRoute's global fetch - // proxy patch interferes with. - const client = new TLSClient(buildNativeTlsClientOptions()) as { - start: () => Promise; - request: (url: string, opts: Record) => Promise; - }; - await client.start(); - - installExitHook(); - return client; - } catch (err) { - clientPromise = null; - const msg = err instanceof Error ? err.message : String(err); - throw new TlsClientUnavailableError( - `TLS impersonation client failed to start: ${msg}. ` + - `Verify tls-client-node is installed and its native binary downloaded.` - ); - } - })(); - } - return clientPromise as Promise<{ - request: (url: string, opts: Record) => Promise; - }>; -} - -interface TlsResponseLike { - status: number; - headers: Record; - body: string; // for non-streaming requests, the full response body - cookies?: Record; - text: () => Promise; - bytes: () => Promise; - json: () => Promise; -} - -export class TlsClientUnavailableError extends Error { - constructor(message: string) { - super(message); - this.name = "TlsClientUnavailableError"; - } -} - -export interface TlsFetchOptions { - method?: "GET" | "POST" | "PUT" | "PATCH" | "DELETE"; - headers?: Record; - body?: string; - timeoutMs?: number; - signal?: AbortSignal | null; - /** - * If true, the response body is streamed to a temp file and exposed as a - * ReadableStream. Use for SSE responses (the perplexity_ask - * endpoint). Otherwise, the full body is read into memory. - */ - stream?: boolean; - /** EOF marker the upstream sends to signal end of stream (default: "[DONE]"). */ - streamEofSymbol?: string; - /** - * Optional upstream proxy URL (`http://user:pass@host:port` or - * `socks5://...`). When set, the request is tunneled through this proxy - * before reaching perplexity.ai. - * - * Resolution order: - * 1. `options.proxyUrl` (per-call override from caller) - * 2. `process.env.OMNIROUTE_TLS_PROXY_URL` (single-flag opt-in) - * 3. `process.env.HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (POSIX-standard fallback) - * - * The native `tls-client-node` binding does **not** consult Go's - * `http.ProxyFromEnvironment`, so the env vars need to be plumbed in here at - * the JS layer. - */ - proxyUrl?: string; -} - -import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; -import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; - -/** - * Resolve the proxy URL for a tls-client request. Per-call value wins; - * otherwise we use the standard proxy fetch resolution which reads from - * the dashboard AsyncLocalStorage context or falls back to env vars. - * - * Fail-closed: if resolution throws (e.g. a configured socks5 proxy with - * ENABLE_SOCKS5_PROXY=false), this rethrows rather than returning undefined — - * undefined would let the native binding connect directly and leak the real IP. - */ -function resolveProxyUrl(perCall: string | undefined): string | undefined { - return resolveTlsClientProxyUrl("https://www.perplexity.ai", perCall, resolveProxyForRequest); -} - -export interface TlsFetchResult { - status: number; - headers: Headers; - /** Full response body as text — only populated for non-streaming requests. */ - text: string | null; - /** Streaming body — only populated when options.stream === true. */ - body: ReadableStream | null; -} - -// Test-only injection point. Tests call __setTlsFetchOverrideForTesting() -// to replace the real TLS client with a mock; production never touches this. -let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = - null; - -export function __setTlsFetchOverrideForTesting(fn: typeof testOverride): void { - testOverride = fn; -} - -/** - * Make a single HTTP request to perplexity.ai with a Firefox-like TLS fingerprint. - * - * Throws TlsClientUnavailableError if the native binary failed to load. - */ -export async function tlsFetchPerplexity( +export const tlsClientModule = createTlsClientModule({ + providerName: "Perplexity", + tlsProfile: "firefox_148", + domain: "https://www.perplexity.ai", + tempDirPrefix: "pplx-stream-", + tailFileVariant: "A", + responseValidation: "sse", + exportCloudflareCheck: true, + defaultTimeoutMs: DEFAULT_TIMEOUT_MS, + hardTimeoutGraceMs: HARD_TIMEOUT_GRACE_MS, +}); + +export const tlsFetchPerplexity = ( url: string, options: TlsFetchOptions = {} -): Promise { - if (testOverride) return testOverride(url, options); - // Honor abort signals up-front. tls-client-node's koffi binding doesn't - // accept an AbortSignal mid-flight (the binary call is opaque), so the best - // we can do is bail before issuing the call. We also re-check after — if - // the caller aborted while the upstream was running, throw rather than - // returning a stale response so the caller doesn't try to use it. - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - const client = await getClient(); - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - - const requestOptions: Record = { - method: options.method || "GET", - headers: options.headers || {}, - body: options.body, - tlsClientIdentifier: PPLX_PROFILE, - timeoutMilliseconds: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, - followRedirects: true, - withRandomTLSExtensionOrder: true, - // Plumb the configured proxy through to the native binding. tls-client-node - // consults `proxyUrl` in the per-call options (it does NOT auto-pick up - // HTTP_PROXY / HTTPS_PROXY env), so callers / env have to be threaded in - // explicitly. See `resolveProxyUrl()` for the lookup order. - proxyUrl: resolveProxyUrl(options.proxyUrl), - }; - - if (options.stream) { - return await tlsFetchStreaming( - client, - url, - requestOptions, - options.streamEofSymbol, - options.signal ?? null, - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS - ); - } - - let tlsResponse: TlsResponseLike; - try { - tlsResponse = await raceWithTimeout( - client.request(url, requestOptions), - (options.timeoutMs ?? DEFAULT_TIMEOUT_MS) + HARD_TIMEOUT_GRACE_MS, - options.signal ?? null - ); - } catch (err) { - if (err instanceof TlsClientHangError) { - // The native binding is wedged — drop the singleton so the next - // request respawns a fresh client (and a fresh koffi load). - resetClientCache(); - } - throw err; - } - if (options.signal?.aborted) { - throw makeAbortError(options.signal); - } - return { - status: tlsResponse.status, - headers: toHeaders(tlsResponse.headers), - text: tlsResponse.body, - body: null, - }; -} - -function makeAbortError(signal: AbortSignal): Error { - const reason = signal.reason; - if (reason instanceof Error) return reason; - const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); - err.name = "AbortError"; - return err; -} - -function toHeaders(raw: Record): Headers { - const h = new Headers(); - for (const [k, vs] of Object.entries(raw || {})) { - for (const v of vs) h.append(k, v); - } - return h; -} - -/** - * Returns true if the response body is a Cloudflare challenge/interstitial page - * rather than a real Perplexity response. From VPS/datacenter IPs a valid cookie - * still gets a 403 "Just a moment..." HTML page; distinguishing it from a genuine - * auth failure lets the caller surface an actionable error (issue #2459). - * - * Exported so the executor and the connection validator share one detector. - */ -export function isCloudflareChallenge(text: string | null | undefined): boolean { - if (!text) return false; - return /just a moment|window\._cf_chl_opt|challenges\.cloudflare\.com|attention required|cf-chl/i.test( - text - ); -} - -// ─── Streaming via temp file ──────────────────────────────────────────────── -// tls-client-node's streaming primitive writes the response body chunk-by-chunk -// to a file path, terminating when the upstream sends `streamOutputEOFSymbol`. -// We tail the file from a worker and surface the bytes as a ReadableStream. - -async function tlsFetchStreaming( - client: { request: (url: string, opts: Record) => Promise }, - url: string, - requestOptions: Record, - eofSymbol = "[DONE]", - signal: AbortSignal | null = null, - hardTimeoutMs: number = DEFAULT_TIMEOUT_MS + HARD_TIMEOUT_GRACE_MS -): Promise { - const dir = await mkdtemp(join(tmpdir(), "pplx-stream-")); - const path = join(dir, `${randomUUID()}.sse`); - - const streamOpts = { - ...requestOptions, - streamOutputPath: path, - streamOutputBlockSize: 1024, - streamOutputEOFSymbol: eofSymbol, - }; - - // Kick off the request without awaiting — tls-client writes the body to - // `path` chunk-by-chunk while the call runs. The Promise resolves when the - // request fully completes (full body written). Wrapping in raceWithTimeout - // guarantees this promise eventually settles even if the koffi binding - // wedges; on hang we reset the singleton so the next request respawns. - let resetOnHang = true; - const requestPromise = raceWithTimeout( - client.request(url, streamOpts), - hardTimeoutMs, - signal - ).catch((err: unknown) => { - if (resetOnHang && err instanceof TlsClientHangError) { - resetClientCache(); - resetOnHang = false; - } - // Re-throw so downstream consumers (waitForContent, tailFile) observe - // the rejection and surface it instead of treating the stream as having - // ended cleanly. - throw err; - }); - - // Wait for the file to exist AND have at least one byte. tls-client-node - // creates the output file when the request starts, but the file can be - // empty for a brief window before the first body chunk lands — peeking - // during that window would return "" and misclassify the response as - // non-SSE, dropping us into the buffered-wait branch and silently turning - // a streaming request into a buffered one. Waiting for content avoids - // that race; if the request actually fails before producing any bytes, - // the timeout falls through to the requestPromise drain below (returning - // the real upstream status). - const ready = await waitForContent(path, 5_000, requestPromise); - if (!ready) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Peek the first bytes to decide whether this looks like SSE. Anything - // that doesn't positively look like SSE (JSON `{...}`, HTML `<...>`, plain - // text rate-limit messages, Cloudflare challenge pages, etc.) gets surfaced - // as a non-streaming response so the executor sees the real upstream status - // and body — otherwise non-2xx error pages get silently treated as 200 OK - // and the SSE parser produces an empty completion. - const peek = await readFirstBytes(path, 256); - if (!looksLikeSse(peek)) { - const r = await requestPromise.catch( - (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike - ); - await cleanupTempPath(path); - return { - status: r.status, - headers: toHeaders(r.headers), - text: r.body, - body: null, - }; - } - - // Looks like SSE — start tailing. SSE bodies in practice are always 2xx; - // tls-client-node doesn't expose response status separately from full-body - // completion, so we report 200 and let the SSE parser consume the stream. - const stream = tailFile(path, eofSymbol, requestPromise, signal); - const headers = new Headers({ - "Content-Type": "text/event-stream", - "Cache-Control": "no-cache", - }); - return { status: 200, headers, text: null, body: stream }; -} - -/** - * Returns true if the peeked response body looks like an SSE stream — i.e., - * begins (after any leading whitespace) with one of the SSE field markers - * (`data:`, `event:`, `id:`, `retry:`) or a comment line (`:`). - * - * Exported for tests. - */ -export function looksLikeSse(text: string): boolean { - const trimmed = text.replace(/^[\s\r\n]+/, ""); - if (!trimmed) return false; - if (trimmed.startsWith(":")) return true; - return /^(data|event|id|retry):/i.test(trimmed); -} - -async function cleanupTempPath(path: string): Promise { - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); -} - -async function readFirstBytes(path: string, n: number): Promise { - const fd = await open(path, "r"); - try { - const buf = Buffer.alloc(n); - const { bytesRead } = await fd.read(buf, 0, n, 0); - return buf.subarray(0, bytesRead).toString("utf8"); - } finally { - await fd.close().catch(() => {}); - } -} - -/** - * Wait for the streaming output file to exist AND contain at least one byte. - * Returns false if the request settles before any bytes arrive (so the caller - * can drain `requestPromise` and surface the real upstream status). Returns - * true as soon as the file has data — even one byte is enough for the SSE - * heuristic to give a useful answer. - */ -async function waitForContent( - path: string, - timeoutMs: number, - requestPromise: Promise -): Promise { - let requestSettled = false; - requestPromise.then( - () => { - requestSettled = true; - }, - () => { - requestSettled = true; - } - ); - const start = Date.now(); - while (Date.now() - start < timeoutMs) { - try { - const s = await stat(path); - if (s.size > 0) return true; - } catch { - // file doesn't exist yet - } - // If the request finished without producing any bytes, no point waiting - // out the rest of the timeout — let the caller drain it. - if (requestSettled) return false; - await sleep(25); - } - return false; -} - -function tailFile( - path: string, - eofSymbol: string, - done: Promise, - signal: AbortSignal | null = null -): ReadableStream { - return new ReadableStream({ - async start(controller) { - const fd = await open(path, "r"); - const buf = Buffer.alloc(64 * 1024); - let offset = 0; - let finished = false; - let aborted = false; - let upstreamError: Error | null = null; - - // Track request settlement, capturing both fulfillment and rejection. - // Without the rejection branch, a mid-stream tls-client-node error - // becomes an unhandledRejection — the stream cleans up silently and - // the consumer sees what looks like a successful truncated response. - done.then( - () => { - finished = true; - }, - (err) => { - upstreamError = err instanceof Error ? err : new Error(String(err)); - finished = true; - } - ); - - // If the caller aborts, stop tailing immediately. - const onAbort = () => { - aborted = true; - }; - if (signal) { - if (signal.aborted) aborted = true; - else signal.addEventListener("abort", onAbort, { once: true }); - } +): Promise => tlsClientModule.tlsFetch(url, options); - let errored = false; - try { - while (!aborted) { - const { bytesRead } = await fd.read(buf, 0, buf.length, offset); - if (bytesRead > 0) { - const chunk = buf.subarray(0, bytesRead); - offset += bytesRead; - const text = chunk.toString("utf8"); - if (text.includes(eofSymbol)) { - const cutAt = text.indexOf(eofSymbol) + eofSymbol.length; - controller.enqueue(new Uint8Array(chunk.subarray(0, cutAt))); - break; - } - controller.enqueue(new Uint8Array(chunk)); - } else if (finished) { - // No more data and request completed. If the request rejected, - // surface the error so the consumer doesn't think the stream - // ended cleanly. - if (upstreamError) { - controller.error(upstreamError); - errored = true; - } - break; - } else { - await sleep(25); - } - } - } catch (err) { - controller.error(err); - errored = true; - } finally { - if (signal) signal.removeEventListener("abort", onAbort); - await fd.close().catch(() => {}); - await unlink(path).catch(() => {}); - const dir = path.substring(0, path.lastIndexOf("/")); - await rmdir(dir).catch(() => {}); - if (!errored) controller.close(); - } - }, - }); -} +export const __setTlsFetchOverrideForTesting = tlsClientModule.__setTlsFetchOverrideForTesting; -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); -} +export { TlsClientHangError, TlsClientUnavailableError } from "./tlsClientBase.ts"; +export type { TlsFetchOptions, TlsFetchResult } from "./tlsClientBase.ts"; +export { looksLikeSse, isCloudflareChallenge } from "./tlsClientBase.ts"; diff --git a/open-sse/services/tlsClientBase.ts b/open-sse/services/tlsClientBase.ts new file mode 100644 index 00000000000..11249864f2c --- /dev/null +++ b/open-sse/services/tlsClientBase.ts @@ -0,0 +1,958 @@ +/** + * Shared TLS client infrastructure — a factory-style base that consolidates + * 6 nearly-identical per-provider TLS client files into one source of truth. + * + * Each provider file calls `createTlsClientModule(config)` to obtain its + * provider-specific `tlsFetch` and `__setTlsFetchOverrideForTesting` exports. + * + * TailFile variants: + * A — Uint8Array enqueue, includes EOF symbol, substring-based cleanup + * ChatGPT, Claude, Perplexity, Notion + * B1 — Buffer.from enqueue, excludes EOF symbol, inline drainRemaining loop + * Grok + * B2 — Buffer.from enqueue, excludes EOF symbol, extracted helpers + * LMArena + * + * Response validation: + * sse — checks `looksLikeSse(peek)`, falls back to buffered + * ChatGPT, Claude, Perplexity, Notion + * cf — checks `isCloudflareChallenge(peek)` → 403, HTML → 502 + * Grok, LMArena + */ + +// --------------------------------------------------------------------------- +// Node imports +// --------------------------------------------------------------------------- +import { tmpdir } from "node:os"; +import { randomUUID } from "node:crypto"; +import { join, dirname } from "node:path"; +import { open, unlink, rmdir, readFile, mkdtemp, stat } from "node:fs/promises"; + +// --------------------------------------------------------------------------- +// Proxy resolution — every provider file imports both of these +// --------------------------------------------------------------------------- +import { resolveProxyForRequest } from "../utils/proxyFetch.ts"; +import { resolveTlsClientProxyUrl } from "./tlsClientProxy.ts"; +import { buildNativeTlsClientOptions } from "./tlsClientDownloadDir.ts"; + +// --------------------------------------------------------------------------- +// Types +// --------------------------------------------------------------------------- + +export interface TlsResponseLike { + status: number; + headers: Record; + body: string; +} + +export interface TlsFetchResult { + status: number; + headers: Headers; + text: string | null; + body: ReadableStream | null; +} + +export interface TlsFetchOptions { + method?: string; + headers?: Record; + body?: string; + signal?: AbortSignal; + timeoutMs?: number; + stream?: boolean; + streamEofSymbol?: string; + byteResponse?: boolean; + proxyUrl?: string; +} + +// --------------------------------------------------------------------------- +// Factory config (one instance per provider stub) +// --------------------------------------------------------------------------- + +export interface TlsClientConfig { + /** Human-readable provider name for logs and error messages. */ + providerName: string; + /** TLS profile identifier (e.g. "chrome_146") */ + tlsProfile: string; + /** Default upstream domain for proxy resolution (e.g. "https://chatgpt.com") */ + domain: string; + /** Temp directory prefix (e.g. "cgpt-stream-") */ + tempDirPrefix: string; + /** EOF symbol for streaming (default "[DONE]") */ + streamEofSymbol?: string; + /** Default timeout in ms (default 60_000) */ + defaultTimeoutMs?: number; + /** Hard timeout grace period in ms (default 10_000) */ + hardTimeoutGraceMs?: number; + /** First-byte timeout for waitForContent (default 5_000; ChatGPT uses 30_000) */ + firstByteTimeoutMs?: number; + /** + * TailFile variant: + * "A" — Uint8Array enqueue, includes EOF, substring cleanup + * "B1" — Buffer.from enqueue, excludes EOF, inline drainRemaining + * "B2" — Buffer.from enqueue, excludes EOF, extracted helpers + */ + tailFileVariant: "A" | "B1" | "B2"; + /** + * Response validation mode: + * "sse" — check looksLikeSse → fall back to buffered + * "cf" — check isCloudflareChallenge → 403, HTML → 502, else stream + */ + responseValidation: "sse" | "cf"; + /** + * Optional override for proxy resolution domain (e.g., LMArena uses + * "https://arena.ai" hardcoded instead of the config domain). + */ + proxyDomainOverride?: string; + /** + * Whether to export `isCloudflareChallenge` from the provider stub. + * Grok, LMArena, Perplexity, Notion all export it. + */ + exportCloudflareCheck: boolean; + /** + * Whether to expose `__tlsFetchStreamingForTesting` (ChatGPT only). + */ + exposeStreamingForTesting?: boolean; +} + +// --------------------------------------------------------------------------- +// Error classes +// --------------------------------------------------------------------------- + +export class TlsClientUnavailableError extends Error { + override name = "TlsClientUnavailableError"; +} + +export class TlsClientHangError extends Error { + override name = "TlsClientHangError"; +} + +// --------------------------------------------------------------------------- +// Shared helpers (identical across all 6 providers) +// --------------------------------------------------------------------------- + +export function sleep(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + +export function makeAbortError(signal: AbortSignal): Error { + const reason = signal.reason; + if (reason instanceof Error) return reason; + const err = new Error(typeof reason === "string" ? reason : "The operation was aborted"); + err.name = "AbortError"; + return err; +} + +export function toHeaders(raw: Record | null | undefined): Headers { + const h = new Headers(); + for (const [k, vs] of Object.entries(raw || {})) { + for (const v of vs) h.append(k, v); + } + return h; +} + +export async function raceWithTimeout( + promise: Promise, + timeoutMs: number, + signal: AbortSignal | null | undefined +): Promise { + // If no signal, just race with a simple timeout. + if (!signal) { + return await Promise.race([ + promise, + new Promise((_, reject) => { + setTimeout(() => reject(new TlsClientHangError()), timeoutMs); + }), + ]); + } + + // With signal, race against both timeout and abort. + return await new Promise((resolve, reject) => { + let settled = false; + + const done = (fn: () => void) => { + if (!settled) { + settled = true; + fn(); + } + }; + + const timer = setTimeout(() => { + done(() => reject(new TlsClientHangError())); + }, timeoutMs); + + const onAbort = () => { + done(() => reject(makeAbortError(signal))); + }; + + if (signal.aborted) { + onAbort(); + } else { + signal.addEventListener("abort", onAbort, { once: true }); + } + + promise.then( + (v) => { + done(() => { + clearTimeout(timer); + signal.removeEventListener("abort", onAbort); + resolve(v); + }); + }, + (e) => { + done(() => { + clearTimeout(timer); + signal.removeEventListener("abort", onAbort); + reject(e); + }); + } + ); + }); +} + +/** Read up to N bytes from a file, returning the utf-8 decoded text. */ +export async function readFirstBytes(path: string, n: number): Promise { + const fd = await open(path, "r"); + try { + const buf = Buffer.alloc(n); + const { bytesRead } = await fd.read(buf, 0, n, 0); + return buf.subarray(0, bytesRead).toString("utf8"); + } finally { + await fd.close().catch(() => {}); + } +} + +/** + * Wait for the streaming output file to exist AND contain at least one byte. + * Returns false if the request settles before any bytes arrive (so the caller + * can drain `requestPromise` and surface the real upstream status). Returns + * true as soon as the file has data. + */ +export async function waitForContent( + path: string, + timeoutMs: number, + requestPromise: Promise +): Promise { + let requestSettled = false; + requestPromise.then( + () => { + requestSettled = true; + }, + () => { + requestSettled = true; + } + ); + const start = Date.now(); + while (Date.now() - start < timeoutMs) { + try { + const s = await stat(path); + if (s.size > 0) return true; + } catch { + // file doesn't exist yet + } + if (requestSettled) return false; + await sleep(25); + } + return false; +} + +/** + * Returns true if the peeked response body looks like an SSE stream — i.e., + * begins (after any leading whitespace) with one of the SSE field markers + * (`data:`, `event:`, `id:`, `retry:`) or a comment line (`:`). + */ +export function looksLikeSse(text: string): boolean { + const trimmed = text.replace(/^[\s\r\n]+/, ""); + if (!trimmed) return false; + if (trimmed.startsWith(":")) return true; + return /^(data|event|id|retry):/i.test(trimmed); +} + +/** + * Returns true if the response body is a Cloudflare challenge/interstitial page. + */ +export function isCloudflareChallenge(text: string | null | undefined): boolean { + if (!text) return false; + return /just a moment|window\._cf_chl_opt|challenges\.cloudflare\.com|attention required|cf-chl/i.test( + text + ); +} + +// --------------------------------------------------------------------------- +// Temp-path cleanup — two variants +// --------------------------------------------------------------------------- + +/** Variant A: substring-based parent dir extraction (ChatGPT, Claude, Perplexity, Notion) */ +async function cleanupTempPathSubstring(path: string): Promise { + await unlink(path).catch(() => {}); + const dir = path.substring(0, path.lastIndexOf("/")); + await rmdir(dir).catch(() => {}); +} + +/** Variant B: dirname-based parent dir extraction (Grok, LMArena) */ +async function cleanupTempPathDirname(path: string): Promise { + await unlink(path).catch(() => {}); + await rmdir(dirname(path)).catch(() => {}); +} + +async function readTextFileIfExists(path: string): Promise { + try { + return await readFile(path, "utf8"); + } catch { + return ""; + } +} + +// --------------------------------------------------------------------------- +// TailFile — Variant A +// Uint8Array enqueue, includes EOF symbol, substring cleanup +// Used by: ChatGPT, Claude, Perplexity, Notion +// --------------------------------------------------------------------------- + +function tailFileVariantA( + path: string, + eofSymbol: string, + done: Promise, + signal: AbortSignal | null = null, + cleanupPath: string +): ReadableStream { + return new ReadableStream({ + async start(controller) { + const fd = await open(path, "r"); + const buf = Buffer.alloc(64 * 1024); + let offset = 0; + let finished = false; + let aborted = false; + let upstreamError: Error | null = null; + + done.then( + () => { + finished = true; + }, + (err) => { + upstreamError = err instanceof Error ? err : new Error(String(err)); + finished = true; + } + ); + + const onAbort = () => { + aborted = true; + }; + if (signal) { + if (signal.aborted) aborted = true; + else signal.addEventListener("abort", onAbort, { once: true }); + } + + let errored = false; + try { + while (!aborted) { + const { bytesRead } = await fd.read(buf, 0, buf.length, offset); + if (bytesRead > 0) { + const chunk = buf.subarray(0, bytesRead); + offset += bytesRead; + const text = chunk.toString("utf8"); + if (text.includes(eofSymbol)) { + const cutAt = text.indexOf(eofSymbol) + eofSymbol.length; + controller.enqueue(new Uint8Array(chunk.subarray(0, cutAt))); + break; + } + controller.enqueue(new Uint8Array(chunk)); + } else if (finished) { + if (upstreamError) { + controller.error(upstreamError); + errored = true; + } + break; + } else { + await sleep(25); + } + } + } catch (err) { + controller.error(err); + errored = true; + } finally { + if (signal) signal.removeEventListener("abort", onAbort); + await fd.close().catch(() => {}); + await cleanupTempPathSubstring(cleanupPath); + if (!errored) controller.close(); + } + }, + }); +} + +// --------------------------------------------------------------------------- +// TailFile — Variant B1 +// Buffer.from enqueue, excludes EOF symbol, inline drainRemaining loop +// Used by: Grok +// --------------------------------------------------------------------------- + +function tailFileVariantB1( + path: string, + eofSymbol: string, + done: Promise, + signal: AbortSignal | null = null, + cleanupPath: string +): ReadableStream { + return new ReadableStream({ + async start(controller) { + const fd = await open(path, "r"); + const buf = Buffer.alloc(64 * 1024); + let offset = 0; + let finished = false; + let aborted = false; + let upstreamError: Error | null = null; + + done.then( + () => { + finished = true; + }, + (err) => { + upstreamError = err instanceof Error ? err : new Error(String(err)); + finished = true; + } + ); + + const onAbort = () => { + aborted = true; + }; + if (signal) { + if (signal.aborted) aborted = true; + else signal.addEventListener("abort", onAbort, { once: true }); + } + + let errored = false; + try { + while (!aborted) { + const { bytesRead } = await fd.read(buf, 0, buf.length, offset); + if (bytesRead > 0) { + const chunk = buf.subarray(0, bytesRead); + offset += bytesRead; + const text = chunk.toString("utf8"); + + if (text.includes(eofSymbol)) { + const beforeEof = text.substring(0, text.indexOf(eofSymbol)); + if (beforeEof) { + controller.enqueue(Buffer.from(beforeEof, "utf8")); + } + controller.close(); + return; + } + + controller.enqueue(Buffer.from(chunk)); + } + + if (finished) { + // Request finished — drain any remaining bytes then close. + while (true) { + const { bytesRead } = await fd.read(buf, 0, buf.length, offset); + if (bytesRead === 0) break; + const chunk = buf.subarray(0, bytesRead); + offset += bytesRead; + const text = chunk.toString("utf8"); + + if (text.includes(eofSymbol)) { + const beforeEof = text.substring(0, text.indexOf(eofSymbol)); + if (beforeEof) { + controller.enqueue(Buffer.from(beforeEof, "utf8")); + } + controller.close(); + return; + } + + controller.enqueue(Buffer.from(chunk)); + } + + if (upstreamError && !errored) { + errored = true; + controller.error(upstreamError); + return; + } + + controller.close(); + return; + } + + await sleep(25); + } + } catch (err) { + if (!errored) { + errored = true; + controller.error(err instanceof Error ? err : new Error(String(err))); + } + } finally { + await fd.close().catch(() => {}); + await cleanupTempPathDirname(cleanupPath); + if (signal) signal.removeEventListener("abort", onAbort); + } + }, + }); +} + +// --------------------------------------------------------------------------- +// TailFile — Variant B2 +// Buffer.from enqueue, excludes EOF symbol, extracted helpers +// Used by: LMArena +// --------------------------------------------------------------------------- + +type FileHandle = Awaited>; + +function enqueueChunkMaybeEof( + controller: ReadableStreamDefaultController, + chunk: Buffer, + eofSymbol: string +): boolean { + const text = chunk.toString("utf8"); + if (!text.includes(eofSymbol)) { + controller.enqueue(Buffer.from(chunk)); + return false; + } + const beforeEof = text.substring(0, text.indexOf(eofSymbol)); + if (beforeEof) controller.enqueue(Buffer.from(beforeEof, "utf8")); + controller.close(); + return true; +} + +async function drainRemaining( + fd: FileHandle, + buf: Buffer, + offsetRef: { offset: number }, + controller: ReadableStreamDefaultController, + eofSymbol: string +): Promise<"closed" | "drained"> { + while (true) { + const { bytesRead } = await fd.read(buf, 0, buf.length, offsetRef.offset); + if (bytesRead === 0) return "drained"; + const chunk = buf.subarray(0, bytesRead); + offsetRef.offset += bytesRead; + if (enqueueChunkMaybeEof(controller, chunk, eofSymbol)) return "closed"; + } +} + +function tailFileVariantB2( + path: string, + eofSymbol: string, + done: Promise, + signal: AbortSignal | null = null, + cleanupPath: string +): ReadableStream { + return new ReadableStream({ + async start(controller) { + const fd = await open(path, "r"); + const buf = Buffer.alloc(64 * 1024); + const offsetRef = { offset: 0 }; + let finished = false; + let aborted = false; + let upstreamError: Error | null = null; + let errored = false; + + done.then( + () => { + finished = true; + }, + (err) => { + upstreamError = err instanceof Error ? err : new Error(String(err)); + finished = true; + } + ); + + const onAbort = () => { + aborted = true; + }; + if (signal) { + if (signal.aborted) aborted = true; + else signal.addEventListener("abort", onAbort, { once: true }); + } + + try { + while (!aborted) { + const { bytesRead } = await fd.read(buf, 0, buf.length, offsetRef.offset); + if (bytesRead > 0) { + const chunk = buf.subarray(0, bytesRead); + offsetRef.offset += bytesRead; + if (enqueueChunkMaybeEof(controller, chunk, eofSymbol)) return; + } + + if (!finished) { + await sleep(25); + continue; + } + + const drained = await drainRemaining(fd, buf, offsetRef, controller, eofSymbol); + if (drained === "closed") return; + if (upstreamError && !errored) { + errored = true; + controller.error(upstreamError); + return; + } + controller.close(); + return; + } + } catch (err) { + if (!errored) { + errored = true; + controller.error(err instanceof Error ? err : new Error(String(err))); + } + } finally { + await fd.close().catch(() => {}); + await cleanupTempPathDirname(cleanupPath); + if (signal) signal.removeEventListener("abort", onAbort); + } + }, + }); +} + +// --------------------------------------------------------------------------- +// Client lifecycle — TLS client singleton per provider +// --------------------------------------------------------------------------- + +/** + * Create a getClient function for a provider stub. + * Uses dynamic `import("tls-client-node")` with `{ runtimeMode: "native" }` + * and `client.start()`, matching the original per-provider lifecycle. + */ +export function createGetClient(config: { + providerName: string; + tlsProfile?: string; +}): () => Promise<{ + request: (url: string, opts: Record) => Promise; +}> { + let clientPromise: Promise<{ + request: (url: string, opts: Record) => Promise; + }> | null = null; + let exitHookInstalled = false; + + const installExitHook = (client: { stop: () => Promise }): void => { + if (!exitHookInstalled) { + exitHookInstalled = true; + process.on("exit", () => { + void client.stop(); + }); + } + }; + + return async function getClient(): Promise<{ + request: (url: string, opts: Record) => Promise; + }> { + if (!clientPromise) { + clientPromise = (async () => { + let TLSClientCtor: { + new (config: Record): { + start: () => Promise; + request: (url: string, opts: Record) => Promise; + stop: () => Promise; + }; + }; + try { + // tls-client-node uses a native binary loaded at runtime. + // The dynamic import delays the binary load until first use — no + // point crashing startup on machines where it's not installed. + const mod = await import("tls-client-node"); + TLSClientCtor = mod.TLSClient; + } catch { + throw new TlsClientUnavailableError( + `tls-client-node is not installed — cannot start TLS client for ${config.providerName}` + ); + } + const tlsOptions: Record = { + ...buildNativeTlsClientOptions(), + }; + if (config.tlsProfile) { + tlsOptions.clientIdentifier = config.tlsProfile; + } + const client = new TLSClientCtor(tlsOptions); + // Start the native TLS client binding + await client.start(); + installExitHook(client); + + return client; + })(); + } + return clientPromise; + }; +} + +/** + * Resolve the proxy URL for a tls-client request. Per-call value wins; + * falls back to the provider-specific env var and the dashboard proxy config. + */ +export function resolveProxyUrl(domain: string, perCall: string | undefined): string | undefined { + return resolveTlsClientProxyUrl(domain, perCall, resolveProxyForRequest); +} + +// --------------------------------------------------------------------------- +// Factory — creates provider-specific tlsFetch + helpers +// --------------------------------------------------------------------------- + +const CLEANUP_VARIANTS = { + A: cleanupTempPathSubstring, + B: cleanupTempPathDirname, +} as const; + +const TAIL_FILE_VARIANTS = { + A: tailFileVariantA, + B1: tailFileVariantB1, + B2: tailFileVariantB2, +} as const; + +export interface TlsClientModule { + tlsFetch: (url: string, options: TlsFetchOptions) => Promise; + __setTlsFetchOverrideForTesting: ( + fn: ((url: string, options: TlsFetchOptions) => Promise) | null + ) => void; + isCloudflareChallenge?: (text: string | null | undefined) => boolean; + __tlsFetchStreamingForTesting?: ( + client: { request: (url: string, opts: Record) => Promise }, + url: string, + requestOptions: Record, + eofSymbol?: string, + signal?: AbortSignal | null, + hardTimeoutMs?: number, + firstByteTimeoutMs?: number + ) => Promise; +} + +/** + * Create a provider-specific TLS client module. + * + * Each provider file calls this once at module level and re-exports + * the returned `tlsFetch` (as e.g. `tlsFetchChatGpt`) and + * `__setTlsFetchOverrideForTesting`. + */ +export function createTlsClientModule(config: TlsClientConfig): TlsClientModule { + const { + providerName, + tlsProfile, + domain, + tempDirPrefix, + streamEofSymbol = "[DONE]", + defaultTimeoutMs = 60_000, + hardTimeoutGraceMs = 10_000, + firstByteTimeoutMs = 5_000, + tailFileVariant, + responseValidation, + proxyDomainOverride, + exportCloudflareCheck, + } = config; + + const getClient = createGetClient({ providerName, tlsProfile }); + + function resetClientCache(): void { + // The getClient closure holds clientPromise — by design the only + // reference is inside getClient's closure. After a hang we need + // the next call to spawn a fresh binding. We achieve this by + // clearing the local reference; the module-level tlsFetch will + // re-read via getClient which recreates it. + // Since getClient's clientPromise is a closure variable, we + // re-create getClient itself: + Object.assign(localState, { + getClient: createGetClient({ providerName, tlsProfile }), + }); + // Note: this is safe because only tlsFetch calls getClient. + // A concurrent in-flight call holds its own reference. + } + + const localState: { getClient: typeof getClient } = { getClient }; + + let testOverride: ((url: string, options: TlsFetchOptions) => Promise) | null = + null; + + const tailFileFn = TAIL_FILE_VARIANTS[tailFileVariant]; + + const cleanupFn = tailFileVariant === "A" ? cleanupTempPathSubstring : cleanupTempPathDirname; + + async function tlsFetchStreaming( + client: { request: (url: string, opts: Record) => Promise }, + url: string, + requestOptions: Record, + eofSymbol: string, + signal: AbortSignal | null, + hardTimeoutMs: number, + firstByteMs: number = firstByteTimeoutMs + ): Promise { + const dir = await mkdtemp(join(tmpdir(), tempDirPrefix)); + const path = join(dir, `${randomUUID()}.sse`); + + const streamOpts: Record = { + ...requestOptions, + streamOutputPath: path, + streamOutputBlockSize: 1024, + streamOutputEOFSymbol: eofSymbol, + }; + + let resetOnHang = true; + const requestPromise = raceWithTimeout( + client.request(url, streamOpts), + hardTimeoutMs, + signal + ).catch((err: unknown) => { + if (resetOnHang && err instanceof TlsClientHangError) { + resetClientCache(); + resetOnHang = false; + } + throw err; + }); + + // Wait for the file to exist AND have at least one byte. + const ready = await waitForContent(path, firstByteMs, requestPromise); + if (!ready) { + const r = await requestPromise.catch( + (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike + ); + const fileText = await readTextFileIfExists(path); + await cleanupFn(path); + return { + status: r.status, + headers: toHeaders(r.headers), + text: r.body || fileText, + body: null, + }; + } + + const peek = await readFirstBytes(path, 256); + + if (responseValidation === "cf") { + // Cloudflare challenge check + if (isCloudflareChallenge(peek)) { + await cleanupFn(path); + return { + status: 403, + headers: new Headers({ "Content-Type": "text/html" }), + text: peek, + body: null, + }; + } + // HTML error page check + if (peek.trimStart().startsWith("<")) { + await cleanupFn(path); + return { + status: 502, + headers: new Headers({ "Content-Type": "text/html" }), + text: peek, + body: null, + }; + } + } else { + // SSE validation — if it doesn't look like SSE, return buffered + if (!looksLikeSse(peek)) { + const r = await requestPromise.catch( + (e) => ({ status: 502, headers: {}, body: String(e) }) as TlsResponseLike + ); + const fileText = await readTextFileIfExists(path); + await cleanupFn(path); + return { + status: r.status, + headers: toHeaders(r.headers), + text: r.body || fileText, + body: null, + }; + } + } + + // Looks valid — create streaming response. + const stream = tailFileFn(path, eofSymbol, requestPromise, signal, path); + + const contentType = responseValidation === "cf" ? "application/x-ndjson" : "text/event-stream"; + + const headers = new Headers({ + "Content-Type": contentType, + "Cache-Control": "no-cache", + }); + return { status: 200, headers, text: null, body: stream }; + } + + async function tlsFetch(url: string, options: TlsFetchOptions = {}): Promise { + // Resolve proxyUrl early so test overrides and the real path both see it. + const resolvedProxyUrl = resolveProxyUrl(proxyDomainOverride ?? domain, options.proxyUrl); + if (testOverride) return testOverride(url, { ...options, proxyUrl: resolvedProxyUrl }); + + if (options.signal?.aborted) { + throw makeAbortError(options.signal); + } + const client = await localState.getClient(); + if (options.signal?.aborted) { + throw makeAbortError(options.signal); + } + + const requestOptions: Record = { + method: options.method || "GET", + headers: options.headers || {}, + body: options.body, + tlsClientIdentifier: tlsProfile, + timeoutMilliseconds: options.timeoutMs ?? defaultTimeoutMs, + followRedirects: true, + withRandomTLSExtensionOrder: true, + proxyUrl: resolvedProxyUrl, + }; + + requestOptions.isByteResponse = options.byteResponse === true; + + if (options.stream) { + return await tlsFetchStreaming( + client, + url, + requestOptions, + options.streamEofSymbol || streamEofSymbol, + options.signal ?? null, + (options.timeoutMs ?? defaultTimeoutMs) + hardTimeoutGraceMs, + firstByteTimeoutMs + ); + } + + let tlsResponse: TlsResponseLike; + try { + tlsResponse = await raceWithTimeout( + client.request(url, requestOptions), + (options.timeoutMs ?? defaultTimeoutMs) + hardTimeoutGraceMs, + options.signal ?? null + ); + } catch (err) { + if (err instanceof TlsClientHangError) { + resetClientCache(); + } + throw err; + } + if (options.signal?.aborted) { + throw makeAbortError(options.signal); + } + return { + status: tlsResponse.status, + headers: toHeaders(tlsResponse.headers), + text: tlsResponse.body, + body: null, + }; + } + + const module: TlsClientModule = { + tlsFetch, + __setTlsFetchOverrideForTesting(fn) { + testOverride = fn; + }, + }; + + if (exportCloudflareCheck) { + module.isCloudflareChallenge = isCloudflareChallenge; + } + + if (config.exposeStreamingForTesting) { + module.__tlsFetchStreamingForTesting = ( + client, + url, + requestOptions, + eofSymbol = "[DONE]", + signal = null, + hardTimeoutMs = defaultTimeoutMs + hardTimeoutGraceMs, + firstByteMs = firstByteTimeoutMs + ): Promise => { + return tlsFetchStreaming( + client, + url, + requestOptions, + eofSymbol, + signal, + hardTimeoutMs, + firstByteMs + ); + }; + } + + return module; +} diff --git a/open-sse/translator/index.ts b/open-sse/translator/index.ts index 501d2ffdbc2..0e4dcc23ff0 100644 --- a/open-sse/translator/index.ts +++ b/open-sse/translator/index.ts @@ -756,6 +756,19 @@ export function translateRequest( delete result[RESPONSES_STORE_MARKER]; } + // #7293 follow-up: the pre-translation hoist above normalizes the *source* + // message array, which a target translator can then undo. `claudeToOpenAI` + // pushes `body.system` as a fresh leading system message before appending the + // converted messages, so an already-hoisted system lands at index 1 again; + // a Responses-source request has no `messages` at all until translation, so + // the earlier call is a no-op for it. Re-run on the final outbound array — + // it is the only shape the upstream actually sees. Idempotent: same array + // reference for non-strict providers and already-compliant requests, so + // prompt-cache prefixes stay stable. + if (targetFormat === FORMATS.OPENAI && result.messages && Array.isArray(result.messages)) { + result.messages = hoistLeadingSystemMessage(result.messages, provider); + } + return result; } diff --git a/open-sse/utils/diagnostics.ts b/open-sse/utils/diagnostics.ts index d872599a983..97c6dc2f100 100644 --- a/open-sse/utils/diagnostics.ts +++ b/open-sse/utils/diagnostics.ts @@ -299,8 +299,15 @@ export function detectMalformedNonStream(resp: unknown): MalformedReason | null ) return true; if (Array.isArray(msg?.tool_calls) && (msg.tool_calls as unknown[]).length > 0) return true; + // Reasoning-only completions are real output: a reasoning model that + // exhausts max_tokens on chain-of-thought returns `content: null` with the + // analysis in a reasoning field. Some OpenAI-compatible upstreams (e.g. + // opencode/mimo-v2.5-free via the OpenCode gateway) name it `reasoning` + // rather than `reasoning_content` — missing either variant falsely flagged + // these as empty_choices → 502 (#6623). if (typeof msg?.reasoning_content === "string" && (msg.reasoning_content as string).length > 0) return true; + if (typeof msg?.reasoning === "string" && (msg.reasoning as string).length > 0) return true; return false; }); diff --git a/open-sse/utils/kimiJwt.ts b/open-sse/utils/kimiJwt.ts new file mode 100644 index 00000000000..56ef8371fe2 --- /dev/null +++ b/open-sse/utils/kimiJwt.ts @@ -0,0 +1,51 @@ +export interface KimiJwtPayload { + sub?: string; + iss?: string; + aud?: string[]; + exp?: number; + iat?: number; + region?: string; + space_id?: string; + typ?: string; + membership?: { level?: number }; + [key: string]: unknown; +} + +export function parseKimiJwt(token: string): KimiJwtPayload | null { + if (!token || typeof token !== "string") return null; + const parts = token.trim().split("."); + if (parts.length !== 3) return null; + try { + const payloadJson = Buffer.from(parts[1], "base64url").toString("utf8"); + const payload = JSON.parse(payloadJson); + if (typeof payload !== "object" || payload === null) return null; + return payload as KimiJwtPayload; + } catch { + return null; + } +} + +export function getKimiTokenExpiration(token: string): { + expiresAtSec: number; + issuedAtSec: number; + remainingSec: number; + isExpired: boolean; +} | null { + const payload = parseKimiJwt(token); + if (!payload || typeof payload.exp !== "number") return null; + + const nowSec = Math.floor(Date.now() / 1000); + const remainingSec = payload.exp - nowSec; + return { + expiresAtSec: payload.exp, + issuedAtSec: typeof payload.iat === "number" ? payload.iat : 0, + remainingSec, + isExpired: remainingSec <= 0, + }; +} + +export function isKimiTokenExpiringSoon(token: string, thresholdSec = 240): boolean { + const exp = getKimiTokenExpiration(token); + if (!exp) return false; + return exp.remainingSec <= thresholdSec; +} diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index 4eab8ea7fc1..1a4e9f410ce 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -10,6 +10,7 @@ import { filterUsageForFormat, normalizeUsage as normalizeTokenUsage, sanitizeUsagePayloadForRequest, + type UsageLike, } from "./usageTracking.ts"; import { parseSSELine, @@ -723,7 +724,7 @@ export function createSSEStream(options: StreamOptions = {}) { !clientExpectsResponsesStream && !clientExpectsClaudeStream && !clientExpectsAntigravityStream; let buffer = ""; - let usage: UsageTokenRecord | null = null; + let usage: UsageLike | null = null; /** Passthrough (OpenAI CC shape): saw tool_calls in stream before finish_reason */ let passthroughHasToolCalls = false; /** Passthrough: whether a chunk with non-null finish_reason was seen (#7800) */ @@ -1032,7 +1033,7 @@ export function createSSEStream(options: StreamOptions = {}) { if ( state?.finishReason && isFinishChunk && - !hasValidUsage(itemSanitized.usage) && + !hasValidUsage(itemSanitized.usage as UsageLike) && totalContentLength > 0 ) { const estimated = estimateUsage(body, totalContentLength, sourceFormat); @@ -2281,7 +2282,7 @@ export function createSSEStream(options: StreamOptions = {}) { pushProviderPayload: (payload: unknown) => providerPayloadCollector.push(payload), pushClientPayload: (payload: unknown) => clientPayloadCollector.push(payload), sanitizeUsagePayload: (payload: unknown) => - sanitizeUsagePayloadForRequest(payload, body, clientResponseFormat), + sanitizeUsagePayloadForRequest(payload as UsageLike, body, clientResponseFormat), setPassthroughResponsesId: (value: string) => { passthroughResponsesId = value; }, diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index e79c858b139..11a6f4e779c 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -507,6 +507,31 @@ export function buildStreamErrorChunks( return encodeSseEvent(errorEvent, { includeDone: true }); } +/** + * Synthesized terminal frames for a graceful truncation (#7699): the upstream + * ended without a terminal marker AFTER content was already forwarded to the + * client. Instead of an `event: error` frame (which would discard the partial + * content and report a mid-response failure), emit a clean Claude completion — + * `message_delta` carrying `stop_reason: "max_tokens"` followed by + * `message_stop` — so Anthropic SDK / Claude Code treat the response as a + * budget-limited finish and keep everything already received. + */ +export function buildGracefulTruncationChunks(clientResponseFormat?: string | null): Uint8Array[] { + if (clientResponseFormat !== FORMATS.CLAUDE) return []; + + return [ + ...encodeSseEvent( + { + type: "message_delta", + delta: { stop_reason: "max_tokens", stop_sequence: null }, + usage: { input_tokens: 0, output_tokens: 0 }, + }, + { event: "message_delta" } + ), + ...encodeSseEvent({ type: "message_stop" }, { event: "message_stop" }), + ]; +} + /** * Minimal `writable` half used by `pipeWithDisconnect`. The real writable is * driven entirely by the upstream-piped readable, so the writer only needs an @@ -534,10 +559,13 @@ export function createNoopAbortWritable(): { * - **#7699, no terminal marker.** Scoped to Claude (`/v1/messages`), which is * the issue's real scope: Anthropic's SSE spec permits a mid-stream * `event: error`, and Claude clients treat a stream ending without - * `message_stop` as an error. For every other format (plain OpenAI chat - * completions included) a done-without-recognized-marker close is NOT - * necessarily a drop — many formats have no `[DONE]` equivalent — so - * synthesising an error there would be a false positive. + * `message_stop` as an error. When content already reached the client this is + * NOT a provider failure — the partial response is valid and must be kept — so + * it resolves to a graceful truncation (`stop_reason: max_tokens`). For every + * other format (plain OpenAI chat completions included) a + * done-without-recognized-marker close is NOT necessarily a drop — many + * formats have no `[DONE]` equivalent — so synthesising an error there would + * be a false positive. * * - **#8649, no content at all.** The stream terminated properly and carried no * model output. Unlike the marker case this is not format-dependent: a @@ -547,17 +575,26 @@ export function createNoopAbortWritable(): { * emptiness is legitimate (length / tool_calls / content_filter / max_tokens / * tool_use) are excluded by the watcher. */ -function resolveSilentCloseReason(input: { +type SilentCloseOutcome = { kind: "truncated" } | { kind: "error"; reason: string }; + +function resolveSilentCloseOutcome(input: { bytesWereForwarded: boolean; clientTerminalSeen: boolean; clientResponseFormat?: string | null; contentWatcher: StreamContentWatcher; -}): string | null { +}): SilentCloseOutcome | null { if (!input.bytesWereForwarded) return null; if (!input.clientTerminalSeen) { - if (input.clientResponseFormat === FORMATS.CLAUDE) { - return "Upstream stream ended without a terminal marker"; + if ( + input.clientResponseFormat === FORMATS.CLAUDE && + input.contentWatcher.sawContent() + ) { + // #7699 — upstream dropped after content reached the client on a Claude + // stream. Keep the partial response: emit a clean max_tokens completion + // instead of an error frame so Anthropic SDK / Claude Code don't report + // a mid-response break. + return { kind: "truncated" }; } // #10443: every known path that produces OpenAI chat chunks emits a // terminal — the response translators (gemini/claude/kiro/cursor-to-openai) @@ -569,13 +606,13 @@ function resolveSilentCloseReason(input: { // legitimate end. Guard on sawContent() so the #8649 empty-content // verdict below keeps its more precise shape for content-free closes. if (input.clientResponseFormat === FORMATS.OPENAI && input.contentWatcher.sawContent()) { - return "Upstream stream ended without a terminal marker"; + return { kind: "error", reason: "Upstream stream ended without a terminal marker" }; } } const watcher = input.contentWatcher; if (watcher.sawSseFrame() && !watcher.sawContent() && !watcher.sawLegitEmptyTerminal()) { - return "Provider returned empty content"; + return { kind: "error", reason: "Provider returned empty content" }; } return null; @@ -659,20 +696,35 @@ export function createDisconnectAwareStream(transformStream, streamController) { const { done, value } = await reader.read(); if (done) { contentWatcher.finish(); - const silentCloseReason = resolveSilentCloseReason({ + const silentClose = resolveSilentCloseOutcome({ bytesWereForwarded, clientTerminalSeen, clientResponseFormat: streamController.clientResponseFormat, contentWatcher, }); - if (silentCloseReason) { + if (silentClose?.kind === "truncated") { + // #7699 — the upstream dropped without a terminal marker after + // content reached the client. Keep the partial response: emit a + // clean `max_tokens` completion instead of an error frame so + // Anthropic SDK / Claude Code don't report a mid-response break. + streamController.handleComplete(); + try { + for (const chunk of buildGracefulTruncationChunks( + streamController.clientResponseFormat + )) { + controller.enqueue(chunk); + } + } catch { + // downstream may have closed; stream already marked complete + } + } else if (silentClose) { streamController.handleError( - Object.assign(new Error(silentCloseReason), { statusCode: 502 }) + Object.assign(new Error(silentClose.reason), { statusCode: 502 }) ); try { for (const chunk of buildStreamErrorChunks( - silentCloseReason, + silentClose.reason, 502, streamController.clientResponseFormat )) { diff --git a/package-lock.json b/package-lock.json index 939fdf00715..782587985be 100644 --- a/package-lock.json +++ b/package-lock.json @@ -14,7 +14,8 @@ "packages/browser-pool" ], "dependencies": { - "@aws-sdk/client-bedrock-runtime": "^3.1111.0", + "@atjsh/llmlingua-2": "3.0.0", + "@aws-sdk/client-bedrock-runtime": "^3.1112.0", "@dnd-kit/core": "^6.3.1", "@dnd-kit/sortable": "^10.0.0", "@dnd-kit/utilities": "^3.2.2", @@ -46,7 +47,7 @@ "ink-spinner": "^5.0.0", "ink-text-input": "^6.0.0", "ioredis": "^5.10.1", - "jose": "^6.2.8", + "jose": "^6.2.9", "js-yaml": "^5.3.0", "jsonc-parser": "^3.3.1", "lowdb": "^7.0.1", @@ -57,11 +58,11 @@ "mermaid": "^11.15.0", "monaco-editor": "^0.56.0", "next": "16.3.1", - "next-intl": "^4.13.6", + "next-intl": "^4.13.7", "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", - "onnxruntime-node": "~1.24.3", + "onnxruntime-node": "~1.27.0", "open": "^11.0.1", "ora": "^9.4.1", "parse5": "^8.0.1", @@ -118,9 +119,9 @@ "@vitejs/plugin-react": "^6.0.5", "bun": "1.3.14", "c8": "^12.0.0", - "concurrently": "^10.0.4", + "concurrently": "^10.0.5", "cross-env": "^10.1.0", - "ctrf": "^0.2.1", + "ctrf": "^0.3.0", "dpdm": "^4.3.0", "eslint": "^9.39.4", "eslint-config-next": "16.3.1", @@ -155,7 +156,7 @@ "node": ">=22.22.2 <23 || >=24.0.0 <27" }, "optionalDependencies": { - "@atjsh/llmlingua-2": "2.0.5", + "@atjsh/llmlingua-2": "3.0.0", "better-sqlite3": "^13.0.2", "js-tiktoken": "^1.0.20", "keytar": "^7.9.0", @@ -352,9 +353,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -369,9 +367,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -386,9 +381,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -403,9 +395,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -555,16 +544,16 @@ } }, "node_modules/@atjsh/llmlingua-2": { - "version": "2.0.5", - "resolved": "https://registry.npmjs.org/@atjsh/llmlingua-2/-/llmlingua-2-2.0.5.tgz", - "integrity": "sha512-cXdGUJgx0e2Sui5gYC8kapOhw1HAxwzh9IuYPdqyB+VlP6SL9imIfyB7I4GTCl/iG+BUxaOqSrLqWsWYvDZuVQ==", + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/@atjsh/llmlingua-2/-/llmlingua-2-3.0.0.tgz", + "integrity": "sha512-SpRg3zzjATSTjbJV/3ldzDGba0yFjlcnCZ0x3QPJnrUm13PHCvlhwKlgET+BAM5SHFD3n6BsFTwsZUxDBwOyDw==", "license": "MIT", "optional": true, "dependencies": { "es-toolkit": "^1.38.0" }, "peerDependencies": { - "@huggingface/transformers": "^3.5.2 || ^4.0.0", + "@huggingface/transformers": "^4.2.0", "js-tiktoken": "*" } }, @@ -608,9 +597,9 @@ } }, "node_modules/@aws-sdk/client-bedrock-runtime": { - "version": "3.1111.0", - "resolved": "https://registry.npmjs.org/@aws-sdk/client-bedrock-runtime/-/client-bedrock-runtime-3.1111.0.tgz", - "integrity": "sha512-+HHZEehmRaGo1F7YVACor/xARM+m1j8YloFaXfoWn4TIPRojUQId/wyItH2jXyro8JoL78CRRZSI1Z8StX0ldQ==", + "version": "3.1112.0", + "resolved": "https://registry.npmjs.org/@aws-sdk/client-bedrock-runtime/-/client-bedrock-runtime-3.1112.0.tgz", + "integrity": "sha512-XHcpR1Z0j2oQrk7/U+YHgKqy9aV73CsTU7VwZ09jlrgK8eiX/34DbiRZN6VSaHFqgxti008MpxeQCVi52o9u1g==", "license": "Apache-2.0", "dependencies": { "@aws-sdk/core": "^3.977.8", @@ -618,7 +607,7 @@ "@aws-sdk/eventstream-handler-node": "^3.972.33", "@aws-sdk/middleware-eventstream": "^3.972.28", "@aws-sdk/middleware-websocket": "^3.972.51", - "@aws-sdk/token-providers": "3.1111.0", + "@aws-sdk/token-providers": "3.1112.0", "@aws-sdk/types": "^3.974.4", "@smithy/core": "^3.31.1", "@smithy/fetch-http-handler": "^5.6.13", @@ -630,6 +619,23 @@ "node": ">=20.0.0" } }, + "node_modules/@aws-sdk/client-bedrock-runtime/node_modules/@aws-sdk/token-providers": { + "version": "3.1112.0", + "resolved": "https://registry.npmjs.org/@aws-sdk/token-providers/-/token-providers-3.1112.0.tgz", + "integrity": "sha512-6PJbuH46F+qxL4Dup9ecsj2DD+JkYdF0ziv/ska/fJxIP2/NYIxff5utlPgaeKRVcSIacVNUP5NMRjCuJn92GA==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.977.8", + "@aws-sdk/nested-clients": "^3.997.43", + "@aws-sdk/types": "^3.974.4", + "@smithy/core": "^3.31.1", + "@smithy/types": "^4.16.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, "node_modules/@aws-sdk/client-s3": { "version": "3.1086.0", "resolved": "https://registry.npmjs.org/@aws-sdk/client-s3/-/client-s3-3.1086.0.tgz", @@ -4285,9 +4291,9 @@ "license": "MIT" }, "node_modules/@formatjs/icu-messageformat-parser": { - "version": "3.5.16", - "resolved": "https://registry.npmjs.org/@formatjs/icu-messageformat-parser/-/icu-messageformat-parser-3.5.16.tgz", - "integrity": "sha512-kl6b/4D56gjGZi4ZewSmvXbalHwjOUI5ogEHPZqw42goeXTTrL7/yuPzvdrvr0QigDtvaOeb+UeMf62jks43Yg==", + "version": "3.5.17", + "resolved": "https://registry.npmjs.org/@formatjs/icu-messageformat-parser/-/icu-messageformat-parser-3.5.17.tgz", + "integrity": "sha512-cN9jhVqT7u0K9tix43fhjoUwL0nazyW6zsNIXs2QdPADr+nurPfYyssUiMqcSCGlPcCiqnYVxgSn7zBSuI+5Bg==", "license": "MIT", "dependencies": { "@formatjs/icu-skeleton-parser": "2.1.11" @@ -4528,6 +4534,97 @@ "sharp": "^0.34.5" } }, + "node_modules/@huggingface/transformers/node_modules/global-agent": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-3.0.0.tgz", + "integrity": "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q==", + "license": "BSD-3-Clause", + "dependencies": { + "boolean": "^3.0.1", + "es6-error": "^4.1.1", + "matcher": "^3.0.0", + "roarr": "^2.15.3", + "semver": "^7.3.2", + "serialize-error": "^7.0.1" + }, + "engines": { + "node": ">=10.0" + } + }, + "node_modules/@huggingface/transformers/node_modules/matcher": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/matcher/-/matcher-3.0.0.tgz", + "integrity": "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng==", + "license": "MIT", + "dependencies": { + "escape-string-regexp": "^4.0.0" + }, + "engines": { + "node": ">=10" + } + }, + "node_modules/@huggingface/transformers/node_modules/onnxruntime-common": { + "version": "1.24.3", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.3.tgz", + "integrity": "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==", + "license": "MIT" + }, + "node_modules/@huggingface/transformers/node_modules/onnxruntime-node": { + "version": "1.24.3", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.24.3.tgz", + "integrity": "sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==", + "hasInstallScript": true, + "license": "MIT", + "os": [ + "win32", + "darwin", + "linux" + ], + "dependencies": { + "adm-zip": "^0.5.16", + "global-agent": "^3.0.0", + "onnxruntime-common": "1.24.3" + } + }, + "node_modules/@huggingface/transformers/node_modules/semver": { + "version": "7.8.5", + "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", + "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", + "license": "ISC", + "bin": { + "semver": "bin/semver.js" + }, + "engines": { + "node": ">=10" + } + }, + "node_modules/@huggingface/transformers/node_modules/serialize-error": { + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-7.0.1.tgz", + "integrity": "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw==", + "license": "MIT", + "dependencies": { + "type-fest": "^0.13.1" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, + "node_modules/@huggingface/transformers/node_modules/type-fest": { + "version": "0.13.1", + "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.13.1.tgz", + "integrity": "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg==", + "license": "(MIT OR CC0-1.0)", + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/@humanfs/core": { "version": "0.19.1", "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", @@ -6453,9 +6550,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -6472,9 +6566,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -6491,9 +6582,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -6510,9 +6598,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8394,9 +8479,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8414,9 +8496,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8434,9 +8513,6 @@ "ppc64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8454,9 +8530,6 @@ "riscv64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8474,9 +8547,6 @@ "riscv64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8494,9 +8564,6 @@ "s390x" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8514,9 +8581,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8534,9 +8598,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8730,9 +8791,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8747,9 +8805,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8764,9 +8819,6 @@ "ppc64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8781,9 +8833,6 @@ "riscv64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8798,9 +8847,6 @@ "riscv64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8815,9 +8861,6 @@ "s390x" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8832,9 +8875,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8849,9 +8889,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -11710,9 +11747,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11729,9 +11763,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11748,9 +11779,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11767,9 +11795,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11786,9 +11811,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11805,9 +11827,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -13890,9 +13909,6 @@ "arm" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -13907,9 +13923,6 @@ "arm" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -13924,9 +13937,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -13941,9 +13951,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -13958,9 +13965,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -13975,9 +13979,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -16703,9 +16704,9 @@ } }, "node_modules/concurrently": { - "version": "10.0.4", - "resolved": "https://registry.npmjs.org/concurrently/-/concurrently-10.0.4.tgz", - "integrity": "sha512-trZql+7l/0+WRAsAnEdctr4+iiOS6ZrViI6H8QWcCF9MFS/LT0dKpe8vluB1to6it+OxSI4VospFTIFMW8DJRw==", + "version": "10.0.5", + "resolved": "https://registry.npmjs.org/concurrently/-/concurrently-10.0.5.tgz", + "integrity": "sha512-JaP/CoftUrCcAFW/g//RbgEGwlelnEae6cfBLgH6ZdO6s8jPkn6p9SB9u6pdVxYXoiSnFqseOlHfrEfF82TVOg==", "dev": true, "license": "MIT", "dependencies": { @@ -17092,16 +17093,16 @@ "license": "MIT" }, "node_modules/ctrf": { - "version": "0.2.1", - "resolved": "https://registry.npmjs.org/ctrf/-/ctrf-0.2.1.tgz", - "integrity": "sha512-iUo/eHcM5yG8aBS3Miqce9NNiZCtmVZxPpgmZEJIZ96bubwj7IpZx3IqsDqCH2FZjR71EH2NLtbBhtfzDjpaUg==", + "version": "0.3.0", + "resolved": "https://registry.npmjs.org/ctrf/-/ctrf-0.3.0.tgz", + "integrity": "sha512-2luVgKCF/A/pgMKY54AUdicCNbU+Hy3Bl+xwcp98inASt/0fnoNC/A4Bwh/GO47cIuD5c7SS3MY/DY42dPf4zQ==", "dev": true, "license": "MIT", "dependencies": { "ajv": "8.20.0", "ajv-formats": "3.0.1", "glob": "13.0.6", - "yargs": "18.0.0" + "yargs": "18.1.0" }, "bin": { "ctrf": "dist/cli/cli.js" @@ -17155,7 +17156,7 @@ "node": ">=20" } }, - "node_modules/ctrf/node_modules/string-width": { + "node_modules/ctrf/node_modules/cliui/node_modules/string-width": { "version": "7.2.0", "resolved": "https://registry.npmjs.org/string-width/-/string-width-7.2.0.tgz", "integrity": "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ==", @@ -17173,6 +17174,23 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/ctrf/node_modules/string-width": { + "version": "8.2.2", + "resolved": "https://registry.npmjs.org/string-width/-/string-width-8.2.2.tgz", + "integrity": "sha512-GaPUh5gfdrYzqeVNZvUfT23vYYxXzKYidUcnMtJg/3rxRV63EFZy3k6xfKlmfeJD0176lnUV/Usr3XcwSvFzpg==", + "dev": true, + "license": "MIT", + "dependencies": { + "get-east-asian-width": "^1.5.0", + "strip-ansi": "^7.1.2" + }, + "engines": { + "node": ">=20" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/ctrf/node_modules/wrap-ansi": { "version": "9.0.2", "resolved": "https://registry.npmjs.org/wrap-ansi/-/wrap-ansi-9.0.2.tgz", @@ -17191,17 +17209,35 @@ "url": "https://github.com/chalk/wrap-ansi?sponsor=1" } }, + "node_modules/ctrf/node_modules/wrap-ansi/node_modules/string-width": { + "version": "7.2.0", + "resolved": "https://registry.npmjs.org/string-width/-/string-width-7.2.0.tgz", + "integrity": "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "emoji-regex": "^10.3.0", + "get-east-asian-width": "^1.0.0", + "strip-ansi": "^7.1.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/ctrf/node_modules/yargs": { - "version": "18.0.0", - "resolved": "https://registry.npmjs.org/yargs/-/yargs-18.0.0.tgz", - "integrity": "sha512-4UEqdc2RYGHZc7Doyqkrqiln3p9X2DZVxaGbwhn2pi7MrRagKaOcIKe8L3OxYcbhXLgLFUS3zAYuQjKBQgmuNg==", + "version": "18.1.0", + "resolved": "https://registry.npmjs.org/yargs/-/yargs-18.1.0.tgz", + "integrity": "sha512-2rAgRKu54VsHkqI0/tYkmluGXHD4KW7yZoycuqDQ15QOTnc2VVfy0nN/1eMhnQLO00A+dwtK20xuCnc1YGeUyg==", "dev": true, "license": "MIT", "dependencies": { "cliui": "^9.0.1", "escalade": "^3.1.1", "get-caller-file": "^2.0.5", - "string-width": "^7.2.0", + "string-width": "^8.2.1", "y18n": "^5.0.5", "yargs-parser": "^22.0.0" }, @@ -21278,17 +21314,15 @@ } }, "node_modules/global-agent": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-3.0.0.tgz", - "integrity": "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q==", + "version": "4.1.3", + "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-4.1.3.tgz", + "integrity": "sha512-KUJEViiuFT3I97t+GYMikLPJS2Lfo/S2F+DQuBWzuzaMPnvt5yyZePzArx36fBzpGTxZjIpDbXLeySLgh+k76g==", "license": "BSD-3-Clause", "dependencies": { - "boolean": "^3.0.1", - "es6-error": "^4.1.1", - "matcher": "^3.0.0", - "roarr": "^2.15.3", - "semver": "^7.3.2", - "serialize-error": "^7.0.1" + "globalthis": "^1.0.2", + "matcher": "^4.0.0", + "semver": "^7.3.5", + "serialize-error": "^8.1.0" }, "engines": { "node": ">=10.0" @@ -22701,9 +22735,9 @@ } }, "node_modules/icu-minify": { - "version": "4.13.6", - "resolved": "https://registry.npmjs.org/icu-minify/-/icu-minify-4.13.6.tgz", - "integrity": "sha512-iYZGCJZ+kX6o7GrxpVe2sOSdW86AvEqh8RQBvWeBd9jqmuABsMc2B6xongACfItLOogyIWH6GuBslNHr79OU8Q==", + "version": "4.13.7", + "resolved": "https://registry.npmjs.org/icu-minify/-/icu-minify-4.13.7.tgz", + "integrity": "sha512-X9gLFtipsP4HHbmy9urh+palImTR9P6lyhvmgbP6iym8i0IwhcsS4Z6KMjkEoCU6O16OJT5JIZkd8xfDROYo/A==", "funding": [ { "type": "individual", @@ -23594,13 +23628,13 @@ } }, "node_modules/intl-messageformat": { - "version": "11.2.13", - "resolved": "https://registry.npmjs.org/intl-messageformat/-/intl-messageformat-11.2.13.tgz", - "integrity": "sha512-JaPaE6TIX+TAS5XLhDUh41geLw4QfBHX4s5pW8Km+L9fVC8HzB9yOuhbh4EMR/F1+8C6b9qk4763Cv+LdOG1kg==", + "version": "11.2.14", + "resolved": "https://registry.npmjs.org/intl-messageformat/-/intl-messageformat-11.2.14.tgz", + "integrity": "sha512-9f2VD1HFuxUvMw0RxsaP8WmMns6JRTnsNB/zghTFrp11ZktiXWwVDeZBPQchBKEmo+Gx/ZhxI7Qht7YglFD4PA==", "license": "BSD-3-Clause", "dependencies": { "@formatjs/fast-memoize": "3.1.7", - "@formatjs/icu-messageformat-parser": "3.5.16" + "@formatjs/icu-messageformat-parser": "3.5.17" } }, "node_modules/intl-messageformat/node_modules/@formatjs/fast-memoize": { @@ -24544,9 +24578,9 @@ } }, "node_modules/jose": { - "version": "6.2.8", - "resolved": "https://registry.npmjs.org/jose/-/jose-6.2.8.tgz", - "integrity": "sha512-Bsdjwm3Qsd/P0jR+BHDe3LytDfY7WBq2HmCCLIwuVRHMuEC9ae7/R474GIUdF1NgCyZjzVo/A9DOiOBtXq8ZoQ==", + "version": "6.2.9", + "resolved": "https://registry.npmjs.org/jose/-/jose-6.2.9.tgz", + "integrity": "sha512-XrchZOFZUl/T3vTwRe8XK+cJrGtMF4th1ARnDfwbBXFKThGhlsxEE4Zu03AD/bjJSt/9jT/mxrOCkJWOg77aPA==", "license": "MIT", "funding": { "url": "https://github.com/sponsors/panva" @@ -25568,6 +25602,17 @@ "node": ">= 14" } }, + "node_modules/libxmljs2/node_modules/brace-expansion": { + "version": "2.1.4", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.1.4.tgz", + "integrity": "sha512-hGfVzPxthbf3+2yjg/RBs60cB0FhqBS/zvdV/4wn4/BmN0bNMMHPc4V/BbFieqf1TKAGGAHnY4eSjajCl0f2Xg==", + "dev": true, + "license": "MIT", + "optional": true, + "dependencies": { + "balanced-match": "^1.0.0" + } + }, "node_modules/libxmljs2/node_modules/cacache": { "version": "19.0.1", "resolved": "https://registry.npmjs.org/cacache/-/cacache-19.0.1.tgz", @@ -26706,15 +26751,18 @@ } }, "node_modules/matcher": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/matcher/-/matcher-3.0.0.tgz", - "integrity": "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng==", + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/matcher/-/matcher-4.0.0.tgz", + "integrity": "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ==", "license": "MIT", "dependencies": { "escape-string-regexp": "^4.0.0" }, "engines": { "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" } }, "node_modules/material-symbols": { @@ -28837,9 +28885,9 @@ } }, "node_modules/next-intl": { - "version": "4.13.6", - "resolved": "https://registry.npmjs.org/next-intl/-/next-intl-4.13.6.tgz", - "integrity": "sha512-loS6tjWWkr/IP+EV1yXUm9URB54QmZOp4+ZsMZNmeYxY8IZxLvO2esUegnXIDxj5DpK/4BsxwDGfGhlqodpkCQ==", + "version": "4.13.7", + "resolved": "https://registry.npmjs.org/next-intl/-/next-intl-4.13.7.tgz", + "integrity": "sha512-j7KnGWt4Ih6TnW1x714R8bX3H+DYP25fqLTYTfUzAFXh0Od57WuQYM/Sf58yalvIXhE6y8sBYHlrCFmr0jPy3g==", "funding": [ { "type": "individual", @@ -28850,12 +28898,12 @@ "dependencies": { "@formatjs/intl-localematcher": "^0.8.1", "@parcel/watcher": "^2.4.1", - "@swc/core": "^1.15.2", - "icu-minify": "^4.13.6", + "@swc/core": "~1.15.47", + "icu-minify": "^4.13.7", "negotiator": "^1.0.0", - "next-intl-swc-plugin-extractor": "^4.13.6", + "next-intl-swc-plugin-extractor": "4.13.7", "po-parser": "^2.1.1", - "use-intl": "^4.13.6" + "use-intl": "^4.13.7" }, "peerDependencies": { "next": "^12.0.0 || ^13.0.0 || ^14.0.0 || ^15.0.0 || ^16.0.0", @@ -28868,9 +28916,9 @@ } }, "node_modules/next-intl-swc-plugin-extractor": { - "version": "4.13.6", - "resolved": "https://registry.npmjs.org/next-intl-swc-plugin-extractor/-/next-intl-swc-plugin-extractor-4.13.6.tgz", - "integrity": "sha512-M2L8jtPEAXj0CPmXbiW66THdr3OnDqA9IsU1hqv3CdxtVow3Bl9eXPdT9Opeji7L4AFrUZ016dJOs+CoTw66OA==", + "version": "4.13.7", + "resolved": "https://registry.npmjs.org/next-intl-swc-plugin-extractor/-/next-intl-swc-plugin-extractor-4.13.7.tgz", + "integrity": "sha512-MxOUMGKncc/D6rofu0O80I6Ebr3gqlH8XiEb0Uu5TZmX7DqY3NVHs70Uy4bvtgWmHeDwsra6KU8EBtfRVGRv7Q==", "license": "MIT" }, "node_modules/next-themes": { @@ -29685,15 +29733,15 @@ } }, "node_modules/onnxruntime-common": { - "version": "1.24.3", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.3.tgz", - "integrity": "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==", + "version": "1.27.0", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.27.0.tgz", + "integrity": "sha512-3KxL5wIVqa8Ex08jxSzncm9CMgw8CjOFyOQ7SxvG9o0cVLlhTNKXyIQuTbtX4tGPJEf73OER2xrjt4HJSBL4ow==", "license": "MIT" }, "node_modules/onnxruntime-node": { - "version": "1.24.3", - "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.24.3.tgz", - "integrity": "sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==", + "version": "1.27.0", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.27.0.tgz", + "integrity": "sha512-QEzGwrvNBgv4uPVdnbHsOGG4G6T96mdlcFI8aAKPjMU8wOPpVocPXb6k3QGkaZagVTv2G9Bnnbo6Z3JdXr1fQw==", "hasInstallScript": true, "license": "MIT", "os": [ @@ -29703,8 +29751,8 @@ ], "dependencies": { "adm-zip": "^0.5.16", - "global-agent": "^3.0.0", - "onnxruntime-common": "1.24.3" + "global-agent": "^4.1.3", + "onnxruntime-common": "1.27.0" } }, "node_modules/onnxruntime-web": { @@ -29878,9 +29926,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -29920,9 +29965,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -29936,9 +29978,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -31013,6 +31052,134 @@ "ctrf": "^0.2.0" } }, + "node_modules/playwright-ctrf-json-reporter/node_modules/ajv": { + "version": "8.20.0", + "resolved": "https://registry.npmjs.org/ajv/-/ajv-8.20.0.tgz", + "integrity": "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==", + "dev": true, + "license": "MIT", + "dependencies": { + "fast-deep-equal": "^3.1.3", + "fast-uri": "^3.0.1", + "json-schema-traverse": "^1.0.0", + "require-from-string": "^2.0.2" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/epoberezkin" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/ansi-styles": { + "version": "6.2.3", + "resolved": "https://registry.npmjs.org/ansi-styles/-/ansi-styles-6.2.3.tgz", + "integrity": "sha512-4Dj6M28JB+oAH8kFkTLUo+a2jwOFkuqb3yucU0CANcRRUbxS0cP0nZYCGjcc3BNXwRIsUVmDGgzawme7zvJHvg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=12" + }, + "funding": { + "url": "https://github.com/chalk/ansi-styles?sponsor=1" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/cliui": { + "version": "9.0.1", + "resolved": "https://registry.npmjs.org/cliui/-/cliui-9.0.1.tgz", + "integrity": "sha512-k7ndgKhwoQveBL+/1tqGJYNz097I7WOvwbmmU2AR5+magtbjPWQTS1C5vzGkBC8Ym8UWRzfKUzUUqFLypY4Q+w==", + "dev": true, + "license": "ISC", + "dependencies": { + "string-width": "^7.2.0", + "strip-ansi": "^7.1.0", + "wrap-ansi": "^9.0.0" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/ctrf": { + "version": "0.2.1", + "resolved": "https://registry.npmjs.org/ctrf/-/ctrf-0.2.1.tgz", + "integrity": "sha512-iUo/eHcM5yG8aBS3Miqce9NNiZCtmVZxPpgmZEJIZ96bubwj7IpZx3IqsDqCH2FZjR71EH2NLtbBhtfzDjpaUg==", + "dev": true, + "license": "MIT", + "dependencies": { + "ajv": "8.20.0", + "ajv-formats": "3.0.1", + "glob": "13.0.6", + "yargs": "18.0.0" + }, + "bin": { + "ctrf": "dist/cli/cli.js" + }, + "engines": { + "node": ">=20.19.0" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/string-width": { + "version": "7.2.0", + "resolved": "https://registry.npmjs.org/string-width/-/string-width-7.2.0.tgz", + "integrity": "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "emoji-regex": "^10.3.0", + "get-east-asian-width": "^1.0.0", + "strip-ansi": "^7.1.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/wrap-ansi": { + "version": "9.0.2", + "resolved": "https://registry.npmjs.org/wrap-ansi/-/wrap-ansi-9.0.2.tgz", + "integrity": "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww==", + "dev": true, + "license": "MIT", + "dependencies": { + "ansi-styles": "^6.2.1", + "string-width": "^7.0.0", + "strip-ansi": "^7.1.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "url": "https://github.com/chalk/wrap-ansi?sponsor=1" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/yargs": { + "version": "18.0.0", + "resolved": "https://registry.npmjs.org/yargs/-/yargs-18.0.0.tgz", + "integrity": "sha512-4UEqdc2RYGHZc7Doyqkrqiln3p9X2DZVxaGbwhn2pi7MrRagKaOcIKe8L3OxYcbhXLgLFUS3zAYuQjKBQgmuNg==", + "dev": true, + "license": "MIT", + "dependencies": { + "cliui": "^9.0.1", + "escalade": "^3.1.1", + "get-caller-file": "^2.0.5", + "string-width": "^7.2.0", + "y18n": "^5.0.5", + "yargs-parser": "^22.0.0" + }, + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=23" + } + }, + "node_modules/playwright-ctrf-json-reporter/node_modules/yargs-parser": { + "version": "22.0.0", + "resolved": "https://registry.npmjs.org/yargs-parser/-/yargs-parser-22.0.0.tgz", + "integrity": "sha512-rwu/ClNdSMpkSrUb+d6BRsSkLUq1fmfsY6TOpYzTwvwkg1/NRG85KBy3kq++A8LKQwX6lsu+aWad+2khvuXrqw==", + "dev": true, + "license": "ISC", + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=23" + } + }, "node_modules/playwright-extra": { "version": "4.3.6", "resolved": "https://registry.npmjs.org/playwright-extra/-/playwright-extra-4.3.6.tgz", @@ -33229,12 +33396,6 @@ "node": ">=8.0" } }, - "node_modules/roarr/node_modules/sprintf-js": { - "version": "1.1.3", - "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.1.3.tgz", - "integrity": "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA==", - "license": "BSD-3-Clause" - }, "node_modules/robot3": { "version": "0.4.1", "resolved": "https://registry.npmjs.org/robot3/-/robot3-0.4.1.tgz", @@ -33627,12 +33788,12 @@ } }, "node_modules/serialize-error": { - "version": "7.0.1", - "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-7.0.1.tgz", - "integrity": "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw==", + "version": "8.1.0", + "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-8.1.0.tgz", + "integrity": "sha512-3NnuWfM6vBYoy5gZFvHiYsVbafvI9vZv/+jlIigFn4oP4zjNPK3LhcY0xSCgeb1a5L8jO71Mit9LlNoi2UfDDQ==", "license": "MIT", "dependencies": { - "type-fest": "^0.13.1" + "type-fest": "^0.20.2" }, "engines": { "node": ">=10" @@ -33642,9 +33803,9 @@ } }, "node_modules/serialize-error/node_modules/type-fest": { - "version": "0.13.1", - "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.13.1.tgz", - "integrity": "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg==", + "version": "0.20.2", + "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.20.2.tgz", + "integrity": "sha512-Ne+eE4r0/iWnpAxD852z3A+N0Bt5RN//NjJwRd2VFHEmrywxf5vsZlh4R6lixl6B+wz/8d+maTSAkN1FIkI3LQ==", "license": "(MIT OR CC0-1.0)", "engines": { "node": ">=10" @@ -34413,6 +34574,12 @@ "node": ">= 10.x" } }, + "node_modules/sprintf-js": { + "version": "1.1.3", + "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.1.3.tgz", + "integrity": "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA==", + "license": "BSD-3-Clause" + }, "node_modules/sql.js": { "version": "1.14.2", "resolved": "https://registry.npmjs.org/sql.js/-/sql.js-1.14.2.tgz", @@ -36492,9 +36659,9 @@ } }, "node_modules/use-intl": { - "version": "4.13.6", - "resolved": "https://registry.npmjs.org/use-intl/-/use-intl-4.13.6.tgz", - "integrity": "sha512-RLej84qL6PGTDp/PSG3tqRpwr7IvJfOu4Qfv/uyy8CrnYn1oEOQb6osNJb++jZ7FQxqN3aQ5BI7wTIerUgrgMA==", + "version": "4.13.7", + "resolved": "https://registry.npmjs.org/use-intl/-/use-intl-4.13.7.tgz", + "integrity": "sha512-vWapep/2GESovKEmkxaG1Bkt6AWwANCWrM4kwOYbMmvQ4IsBGJFx9l66h8DNCcU0jSOzNtSFyRXFcYvgAak/Cg==", "funding": [ { "type": "individual", @@ -36505,7 +36672,7 @@ "dependencies": { "@formatjs/fast-memoize": "^3.1.0", "@schummar/icu-type-parser": "1.21.5", - "icu-minify": "^4.13.6", + "icu-minify": "^4.13.7", "intl-messageformat": "^11.1.0" }, "peerDependencies": { diff --git a/package.json b/package.json index 5dcd258e8b7..6caf3dc373c 100644 --- a/package.json +++ b/package.json @@ -261,7 +261,7 @@ "alibaba:sync-allowlist": "node --import tsx/esm scripts/ops/sync-alibaba-allowlist.mjs" }, "dependencies": { - "@aws-sdk/client-bedrock-runtime": "^3.1111.0", + "@aws-sdk/client-bedrock-runtime": "^3.1112.0", "@dnd-kit/core": "^6.3.1", "@dnd-kit/sortable": "^10.0.0", "@dnd-kit/utilities": "^3.2.2", @@ -292,7 +292,7 @@ "ink-spinner": "^5.0.0", "ink-text-input": "^6.0.0", "ioredis": "^5.10.1", - "jose": "^6.2.8", + "jose": "^6.2.9", "js-yaml": "^5.3.0", "jsonc-parser": "^3.3.1", "lowdb": "^7.0.1", @@ -303,7 +303,7 @@ "mermaid": "^11.15.0", "monaco-editor": "^0.56.0", "next": "16.3.1", - "next-intl": "^4.13.6", + "next-intl": "^4.13.7", "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", @@ -339,10 +339,10 @@ "zod": "^4.4.3", "zustand": "^5.0.15", "@huggingface/transformers": "^4.2.0", - "onnxruntime-node": "~1.24.3" + "onnxruntime-node": "~1.27.0" }, "optionalDependencies": { - "@atjsh/llmlingua-2": "2.0.5", + "@atjsh/llmlingua-2": "3.0.0", "better-sqlite3": "^13.0.2", "js-tiktoken": "^1.0.20", "keytar": "^7.9.0", @@ -370,9 +370,9 @@ "@vitejs/plugin-react": "^6.0.5", "bun": "1.3.14", "c8": "^12.0.0", - "concurrently": "^10.0.4", + "concurrently": "^10.0.5", "cross-env": "^10.1.0", - "ctrf": "^0.2.1", + "ctrf": "^0.3.0", "dpdm": "^4.3.0", "eslint": "^9.39.4", "eslint-config-next": "16.3.1", diff --git a/scripts/ad-hoc/dump-auto-combos.ts b/scripts/ad-hoc/dump-auto-combos.ts new file mode 100644 index 00000000000..0e3c3520115 --- /dev/null +++ b/scripts/ad-hoc/dump-auto-combos.ts @@ -0,0 +1,52 @@ +/** + * One-shot diagnostic: resolve every built-in auto-combo template and dump the + * resulting candidate pool, weight pack, and config as JSON for inspection. + * + * Run from repo root: + * node --import tsx/esm scripts/ad-hoc/dump-auto-combos.ts > _tasks/research/auto-combos-snapshot.json + */ + +const { AUTO_TEMPLATE_VARIANTS, AUTO_SUFFIX_VARIANTS, AUTO_FAMILY_IDS } = + await import("@omniroute/open-sse/services/autoCombo/builtinCatalog"); +const { createBuiltinAutoCombo, prepareBuiltinAutoComboInputs } = + await import("@omniroute/open-sse/services/autoCombo/builtinCatalog"); + +// Prepares the candidate pool once (DB reads: connections, settings, capabilities) +const prepared = await prepareBuiltinAutoComboInputs(); + +const allTemplates: string[] = []; +allTemplates.push(...Object.keys(AUTO_TEMPLATE_VARIANTS)); +allTemplates.push(...AUTO_SUFFIX_VARIANTS); +allTemplates.push(...AUTO_FAMILY_IDS); + +const results: Array<{ + template: string; + candidateCount: number; + models: string[]; + weightPack: Record; + explorationRate: number; +}> = []; + +for (const name of allTemplates) { + try { + const suffix = name.slice("auto/".length); + const combo = await createBuiltinAutoCombo(name, suffix, prepared as never); + results.push({ + template: name, + candidateCount: combo.models.length, + models: combo.models.map((m) => m.model ?? `${m.providerId}/unknown`), + weightPack: combo.weights ?? {}, + explorationRate: combo.explorationRate, + }); + } catch (err) { + results.push({ + template: name, + candidateCount: 0, + models: [], + weightPack: {}, + explorationRate: 0, + }); + } +} + +console.log(JSON.stringify(results, null, 2)); diff --git a/scripts/build/assembleStandalone.mjs b/scripts/build/assembleStandalone.mjs index b4c8d12c1fa..ee8d730ccf2 100644 --- a/scripts/build/assembleStandalone.mjs +++ b/scripts/build/assembleStandalone.mjs @@ -628,12 +628,11 @@ function copyNativeAssetsAndExtraModules(projectRoot, resolvedOutDir) { * This keeps the fix narrowly scoped to packages the standalone already expects. * * @param {string} projectRoot - * @param {string} resolvedOutDir + * @param {string} bundleNodeModules * @returns {{repaired: number, packages: string[]}} */ -function repairEmptyExternalPackageDirs(projectRoot, resolvedOutDir) { +function repairEmptyExternalPackageDirs(projectRoot, bundleNodeModules) { const summary = { repaired: 0, packages: [] }; - const bundleNodeModules = path.join(resolvedOutDir, "node_modules"); const sourceNodeModules = path.join(projectRoot, "node_modules"); if (!fsSync.existsSync(bundleNodeModules) || !fsSync.existsSync(sourceNodeModules)) { return summary; @@ -899,12 +898,23 @@ export function assembleStandalone({ // 6. Optionally copy native assets + extra modules (synchronous) if (copyNatives) { copyNativeAssetsAndExtraModules(projectRoot, resolvedOutDir); - const emptyPkgRepair = repairEmptyExternalPackageDirs(projectRoot, resolvedOutDir); - if (emptyPkgRepair.repaired > 0) { - console.log( - `[assembleStandalone] Repaired ${emptyPkgRepair.repaired} hollow external package dir(s): ` + - emptyPkgRepair.packages.join(", ") - ); + // Repair hollow externalized package dirs in BOTH locations Turbopack's standalone + // tracer can populate: the top-level bundle node_modules, and — for projects with a + // custom distDir (see next.config.mjs) — the nested /node_modules mirrored + // alongside the traced server chunks. materializeBundledSymlinks (step 7 below) already + // treats these as two distinct targets; #9913 only covered the top-level one, which left + // the nested location's hollow dirs unrepaired (#7346). + for (const bundleNodeModules of [ + path.join(resolvedOutDir, "node_modules"), + path.join(resolvedOutDir, relDistDir, "node_modules"), + ]) { + const emptyPkgRepair = repairEmptyExternalPackageDirs(projectRoot, bundleNodeModules); + if (emptyPkgRepair.repaired > 0) { + console.log( + `[assembleStandalone] Repaired ${emptyPkgRepair.repaired} hollow external package dir(s) in ` + + `${path.relative(resolvedOutDir, bundleNodeModules) || "."}: ${emptyPkgRepair.packages.join(", ")}` + ); + } } // #9166: dynamically imported LLMLingua packages are not reliably traced diff --git a/scripts/build/colocate-standalone.mjs b/scripts/build/colocate-standalone.mjs index 0748cf6db78..bb108da47f9 100644 --- a/scripts/build/colocate-standalone.mjs +++ b/scripts/build/colocate-standalone.mjs @@ -16,14 +16,20 @@ * * Run manually after a build, or automatically via the `postbuild` npm hook. */ -import { cpSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { cpSync, existsSync, mkdirSync, writeFileSync } from "node:fs"; import { dirname, join } from "node:path"; import { execFileSync } from "node:child_process"; -import { fileURLToPath } from "node:url"; +import { fileURLToPath, pathToFileURL } from "node:url"; import { computeDependencyClosure } from "./colocateOptionals.mjs"; const ROOT = dirname(dirname(dirname(fileURLToPath(import.meta.url)))); -const STANDALONE = join(ROOT, ".build", "next", "standalone"); +// STANDALONE defaults to the real build output; OMNIROUTE_STANDALONE_DIR overrides +// it so tests can drive the co-location logic against a synthetic tree without a +// full `next build`. Mirrors the OMNIROUTE_* override seams in the sibling build +// scripts (write-build-sha.mjs, write-build-base-path.mjs, optionalPackStaging.mjs). +const STANDALONE = process.env.OMNIROUTE_STANDALONE_DIR + ? process.env.OMNIROUTE_STANDALONE_DIR + : join(ROOT, ".build", "next", "standalone"); const CALL_LOG_WORKER_REL = join("src", "lib", "usage", "callLogArtifactWorker.js"); const CALL_LOG_WORKER_SRC = join(ROOT, "src", "lib", "usage", "callLogArtifactWorker.ts"); @@ -35,97 +41,138 @@ const WORKER_REL = join( "llmlingua", "onnxWorker.js" ); -const GATE_PKG = join("node_modules", "@atjsh", "llmlingua-2", "package.json"); -const hasOptionals = existsSync( - join(ROOT, "node_modules", "@atjsh", "llmlingua-2", "package.json") -); - -if (!existsSync(STANDALONE)) { - console.log("[colocate-standalone] .build/next/standalone not found — nothing to do."); - process.exit(0); +/** + * Give each esbuild'd ESM worker its OWN `"type":"module"` scope. + * + * The worker bundles are emitted with `--format=esm` under `.js` names, so Node + * needs a nearest-ancestor package.json declaring `"type":"module"` to load them + * as ESM. It is tempting to set that on the standalone ROOT package.json, but the + * standalone entrypoint `server.js` is CommonJS (`require()`, `__dirname`); a root + * `"type":"module"` makes Node parse server.js as ESM and it crashes at startup + * with `ReferenceError: require is not defined in ES module scope`. + * assembleStandalone.mjs::patchStandalonePackageJson strips `type` for exactly + * this reason — re-adding it on the root here reintroduced that crash. + * + * Node resolves module type from the NEAREST package.json, so a scoped + * `{"type":"module"}` beside each worker makes the worker ESM while the root stays + * CommonJS for server.js. Both coexist with no format change and no root edit. + * + * @param {string[]} workerDirs Absolute directories that hold an ESM worker bundle. + * @returns {string[]} The package.json paths that were written (existing ones are left intact). + */ +export function writeEsmWorkerScopes(workerDirs) { + const written = []; + for (const dir of workerDirs) { + const scopedPkgPath = join(dir, "package.json"); + if (existsSync(scopedPkgPath)) continue; // never clobber a traced package.json + try { + writeFileSync(scopedPkgPath, JSON.stringify({ type: "module" }, null, 2) + "\n", "utf8"); + written.push(scopedPkgPath); + console.log(`[colocate-standalone] ✅ ESM scope written: ${scopedPkgPath}`); + } catch (err) { + console.warn(`[colocate-standalone] ⚠️ could not write ESM scope for ${dir}:`, err.message); + } + } + return written; } -const callLogWorkerDest = join(STANDALONE, CALL_LOG_WORKER_REL); -mkdirSync(dirname(callLogWorkerDest), { recursive: true }); -execFileSync( - join(ROOT, "node_modules", ".bin", "esbuild"), - [ - CALL_LOG_WORKER_SRC, - "--bundle", - "--platform=node", - "--packages=external", - "--format=esm", - `--outfile=${callLogWorkerDest}`, - ], - { stdio: "inherit" } -); -console.log("[colocate-standalone] ✅ call-log artifact worker bundled"); -if (!hasOptionals) { - console.log( - "[colocate-standalone] optional SLM deps absent at root node_modules — LLMLingua stays fail-open (slim install)." +function main() { + const hasOptionals = existsSync( + join(ROOT, "node_modules", "@atjsh", "llmlingua-2", "package.json") ); - process.exit(0); -} -// 1) Bundle the worker the resolver expects: /open-sse/.../onnxWorker.js -const workerDest = join(STANDALONE, WORKER_REL); -if (!existsSync(workerDest)) { - mkdirSync(dirname(workerDest), { recursive: true }); - try { - execFileSync( - join(ROOT, "node_modules", ".bin", "esbuild"), - [ - join(ROOT, "open-sse", "services", "compression", "engines", "llmlingua", "onnxWorker.ts"), - "--bundle", - "--platform=node", - "--packages=external", - "--format=esm", - `--outfile=${workerDest}`, - ], - { stdio: "inherit" } - ); - console.log("[colocate-standalone] ✅ LLMLingua worker bundled into standalone tree"); - } catch (err) { - console.warn("[colocate-standalone] ⚠️ worker bundle error:", err.message); + if (!existsSync(STANDALONE)) { + console.log("[colocate-standalone] .build/next/standalone not found — nothing to do."); + return; } -} else { - console.log("[colocate-standalone] worker already present (skipping bundle)"); -} -// 2) Co-locate the optional-dep closure (NO-CLOBBER, same semantics as colocateOptionals.mjs) -const srcNm = join(ROOT, "node_modules"); -const dstNm = join(STANDALONE, "node_modules"); -const closure = computeDependencyClosure(srcNm); -let copied = 0; -for (const pkg of closure) { - const src = join(srcNm, pkg); - const dst = join(dstNm, pkg); - if (!existsSync(src)) continue; - if (existsSync(dst)) continue; // no-clobber: keep traced instances (e.g. pinned @huggingface/transformers) - mkdirSync(dirname(dst), { recursive: true }); - cpSync(src, dst, { recursive: true }); - copied++; -} -console.log( - `[colocate-standalone] ✅ optional-dep closure: ${closure.length} packages (copied ${copied})` -); + const callLogWorkerDest = join(STANDALONE, CALL_LOG_WORKER_REL); + mkdirSync(dirname(callLogWorkerDest), { recursive: true }); + execFileSync( + join(ROOT, "node_modules", ".bin", "esbuild"), + [ + CALL_LOG_WORKER_SRC, + "--bundle", + "--platform=node", + "--packages=external", + "--format=esm", + `--outfile=${callLogWorkerDest}`, + ], + { stdio: "inherit" } + ); + console.log("[colocate-standalone] ✅ call-log artifact worker bundled"); -// 3) Ensure standalone package.json declares "type": "module" so Node 24 runs ESM worker bundles without warning -const standalonePkgPath = join(STANDALONE, "package.json"); -if (existsSync(standalonePkgPath)) { - try { - const rawPkg = readFileSync(standalonePkgPath, "utf8"); - const pkgJson = JSON.parse(rawPkg); - if (!pkgJson.type) { - pkgJson.type = "module"; - writeFileSync(standalonePkgPath, JSON.stringify(pkgJson, null, 2) + "\n", "utf8"); - console.log("[colocate-standalone] ✅ standalone package.json configured with type: module"); - } - } catch (err) { - console.warn( - "[colocate-standalone] ⚠️ could not update standalone package.json:", - err.message + // The call-log worker is always present; scope it to ESM immediately. The + // optional LLMLingua worker dir is added below only when its deps are installed. + const workerDirs = [dirname(callLogWorkerDest)]; + + if (!hasOptionals) { + console.log( + "[colocate-standalone] optional SLM deps absent at root node_modules — LLMLingua stays fail-open (slim install)." ); + writeEsmWorkerScopes(workerDirs); + return; + } + + // 1) Bundle the worker the resolver expects: /open-sse/.../onnxWorker.js + const workerDest = join(STANDALONE, WORKER_REL); + if (!existsSync(workerDest)) { + mkdirSync(dirname(workerDest), { recursive: true }); + try { + execFileSync( + join(ROOT, "node_modules", ".bin", "esbuild"), + [ + join( + ROOT, + "open-sse", + "services", + "compression", + "engines", + "llmlingua", + "onnxWorker.ts" + ), + "--bundle", + "--platform=node", + "--packages=external", + "--format=esm", + `--outfile=${workerDest}`, + ], + { stdio: "inherit" } + ); + console.log("[colocate-standalone] ✅ LLMLingua worker bundled into standalone tree"); + } catch (err) { + console.warn("[colocate-standalone] ⚠️ worker bundle error:", err.message); + } + } else { + console.log("[colocate-standalone] worker already present (skipping bundle)"); } + workerDirs.push(dirname(workerDest)); + + // 2) Co-locate the optional-dep closure (NO-CLOBBER, same semantics as colocateOptionals.mjs) + const srcNm = join(ROOT, "node_modules"); + const dstNm = join(STANDALONE, "node_modules"); + const closure = computeDependencyClosure(srcNm); + let copied = 0; + for (const pkg of closure) { + const src = join(srcNm, pkg); + const dst = join(dstNm, pkg); + if (!existsSync(src)) continue; + if (existsSync(dst)) continue; // no-clobber: keep traced instances (e.g. pinned @huggingface/transformers) + mkdirSync(dirname(dst), { recursive: true }); + cpSync(src, dst, { recursive: true }); + copied++; + } + console.log( + `[colocate-standalone] ✅ optional-dep closure: ${closure.length} packages (copied ${copied})` + ); + + // 3) Give each esbuild'd ESM worker its own "type":"module" scope (see helper doc). + writeEsmWorkerScopes(workerDirs); +} + +// Run as a script (npm `postbuild` hook), but stay importable for unit tests. +const entryScript = process.argv[1] ? pathToFileURL(process.argv[1]).href : null; +if (entryScript === import.meta.url) { + main(); } diff --git a/scripts/check/check-env-doc-sync.mjs b/scripts/check/check-env-doc-sync.mjs index 097b24c60eb..f2959131a01 100644 --- a/scripts/check/check-env-doc-sync.mjs +++ b/scripts/check/check-env-doc-sync.mjs @@ -205,6 +205,10 @@ const IGNORE_FROM_CODE = new Set([ // NVIDIA diagnostic/test helpers used only by ad-hoc scripts. "NVIDIA_BASE_URL", "NVIDIA_MODEL", + // Discord integration ad-hoc script (scripts/ad-hoc/mesh-send.mjs) — + // operator-supplied bot credentials, not user-facing OmniRoute config. + "BOT_TOKEN", + "BOT_URL", // XDG standard data directory — set by OS/desktop session, not OmniRoute config. // Read by setup-open-code.mjs to locate platform-specific OpenCode data dir. "XDG_DATA_HOME", diff --git a/scripts/check/check-public-creds.mjs b/scripts/check/check-public-creds.mjs index 7e065705d0a..21fbe3b563e 100644 --- a/scripts/check/check-public-creds.mjs +++ b/scripts/check/check-public-creds.mjs @@ -98,6 +98,7 @@ export const KNOWN_LITERAL_CREDS = new Set([ "open-sse/services/usage/minimax.ts:213:minimax", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (getMiniMaxUsage signature) "open-sse/services/usage/minimax.ts:213:minimax-cn", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (getMiniMaxUsage signature) "open-sse/executors/zcodeProtocol.ts:302:omniroute-${process.pid}", // local per-process ZCode handshake ID, not an upstream credential + "open-sse/executors/copilot-m365-web.ts:330:access_token=${result.accessToken}; chathubPath=${chathubPath}", // dynamic header format string in token refresh ]); /** diff --git a/scripts/dev/smoke-electron-packaged.mjs b/scripts/dev/smoke-electron-packaged.mjs index 03717912ff7..72afc2f4a7c 100644 --- a/scripts/dev/smoke-electron-packaged.mjs +++ b/scripts/dev/smoke-electron-packaged.mjs @@ -409,45 +409,115 @@ async function settleAfterReady({ getExitState, logs, settleMs }) { } } -async function main() { - const appExecutable = discoverPackagedExecutable(); - if (!existsSync(appExecutable)) { +function assertExecutableExists(appExecutable) { + if (existsSync(appExecutable)) return; + + throw new Error( + `Packaged OmniRoute executable not found at ${appExecutable}. Build it first with \`npm run build: --prefix electron\` or set ELECTRON_SMOKE_APP_EXECUTABLE.` + ); +} + +// ── CI sandbox workaround ────────────────────────────────── +// GitHub Actions runners cannot set SUID on chrome-sandbox (Linux) +// and Windows runners may fail silently without --no-sandbox. +function buildCiSpawnArgs(currentPlatform = platform()) { + if (!process.env.CI) return []; + + const spawnArgs = ["--no-sandbox", "--disable-gpu"]; + if (currentPlatform === "linux") { + spawnArgs.push("--disable-dev-shm-usage"); + } + return spawnArgs; +} + +const NATIVE_DRIVER_LOG_PATTERN = /\[DB\] Driver: (bun:sqlite|better-sqlite3|node:sqlite) \|/; +const SQLJS_DRIVER_LOG_PATTERN = /\[DB\] Driver: sql\.js \|/; + +/** + * Regression guard for #7592: on a packaged app's SECOND launch against an + * already-persisted DATA_DIR, a stale-ABI better-sqlite3 binary (resolved via + * a Turbopack-hashed import) used to fail to load and silently fall through + * to the sql.js (WASM) driver — which then OOMs/retry-loops on real-sized + * databases. Asserts the startup log shows a native driver was selected. + */ +export function assertNativeDriverSelected(logs) { + if (NATIVE_DRIVER_LOG_PATTERN.test(logs)) return; + + if (SQLJS_DRIVER_LOG_PATTERN.test(logs)) { throw new Error( - `Packaged OmniRoute executable not found at ${appExecutable}. Build it first with \`npm run build: --prefix electron\` or set ELECTRON_SMOKE_APP_EXECUTABLE.` + "Packaged Electron app fell back to the sql.js (WASM) driver instead of a native SQLite " + + "driver — this is the regression #7592 guards against (stale-ABI better-sqlite3 binary)." ); } - const smokeUrl = process.env.ELECTRON_SMOKE_URL || DEFAULT_URL; - const timeoutMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_TIMEOUT_MS, DEFAULT_TIMEOUT_MS); - const settleMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_SETTLE_MS, DEFAULT_SETTLE_MS); - const dataDir = - process.env.ELECTRON_SMOKE_DATA_DIR || - (await mkdtemp(join(tmpdir(), "omniroute-electron-smoke-"))); - const removeDataDir = - !process.env.ELECTRON_SMOKE_DATA_DIR && process.env.ELECTRON_SMOKE_KEEP_DATA !== "1"; - const smokeEnv = buildSmokeEnv({ dataDir }); + throw new Error( + "Packaged Electron app logs contain no '[DB] Driver: ...' line — cannot confirm which SQLite " + + "driver loaded." + ); +} - await assertPortIsFree(smokeUrl); - await ensureSmokeEnvDirs(smokeEnv, dataDir); +async function waitForReady({ logs, smokeUrl, timeoutMs, settleMs, exitState }) { + const startedAt = Date.now(); + let lastError = null; - // ── CI sandbox workaround ────────────────────────────────── - // GitHub Actions runners cannot set SUID on chrome-sandbox (Linux) - // and Windows runners may fail silently without --no-sandbox. - const spawnArgs = []; - if (process.env.CI) { - spawnArgs.push("--no-sandbox", "--disable-gpu"); - if (platform() === "linux") { - spawnArgs.push("--disable-dev-shm-usage"); + while (Date.now() - startedAt < timeoutMs) { + assertNoFatalLogs(logs.value); + + if (exitState.spawnError !== null) { + throw new Error(`Packaged Electron app failed to launch: ${exitState.spawnError.message}`); } + if (exitState.exitCode !== null || exitState.signalCode !== null) { + throw new Error( + `Packaged Electron app exited before readiness: code=${exitState.exitCode} signal=${exitState.signalCode}` + ); + } + + try { + const response = await fetchWithTimeout(smokeUrl, 1_000); + if (response.status === 200) { + assertNoFatalLogs(logs.value); + console.log(`[electron-smoke] ready: ${smokeUrl} returned HTTP 200`); + await settleAfterReady({ + getExitState: () => ({ exitCode: exitState.exitCode, signalCode: exitState.signalCode }), + logs, + settleMs, + }); + console.log(`[electron-smoke] stable for ${settleMs}ms after readiness`); + return; + } + lastError = new Error(`HTTP ${response.status}`); + } catch (error) { + lastError = error; + } + + await sleep(500); } + throw new Error( + `Packaged Electron app did not serve ${smokeUrl} within ${timeoutMs}ms. Last error: ${ + lastError instanceof Error ? lastError.message : String(lastError) + }` + ); +} + +/** + * Launches the packaged app once against `dataDir`, waits for readiness + + * settle, tears it down, and returns the captured stdout/stderr text. Shared + * by the single-launch path and the cold-restart (two-launch) path so both + * exercise identical spawn/readiness/shutdown behavior. + */ +async function launchAndCollectLogs({ appExecutable, smokeUrl, dataDir, timeoutMs, settleMs, streamLogs }) { + const smokeEnv = buildSmokeEnv({ dataDir }); + await assertPortIsFree(smokeUrl); + await ensureSmokeEnvDirs(smokeEnv, dataDir); + + const spawnArgs = buildCiSpawnArgs(); console.log(`[electron-smoke] launching ${appExecutable}`); if (spawnArgs.length) console.log(`[electron-smoke] CI args: ${spawnArgs.join(" ")}`); console.log(`[electron-smoke] DATA_DIR=${dataDir}`); console.log(`[electron-smoke] waiting for ${smokeUrl}`); const logs = { value: "" }; - const streamLogs = process.env.ELECTRON_SMOKE_STREAM_LOGS === "1"; const child = spawn(appExecutable, spawnArgs, { detached: platform() !== "win32", env: smokeEnv, @@ -457,60 +527,18 @@ async function main() { child.stdout?.on("data", (chunk) => appendLog(logs, chunk, "[electron] ", streamLogs)); child.stderr?.on("data", (chunk) => appendLog(logs, chunk, "[electron:err] ", streamLogs)); - let exitCode = null; - let signalCode = null; - let spawnError = null; + const exitState = { exitCode: null, signalCode: null, spawnError: null }; child.once("exit", (code, signal) => { - exitCode = code; - signalCode = signal; + exitState.exitCode = code; + exitState.signalCode = signal; }); child.once("error", (error) => { - spawnError = error; + exitState.spawnError = error; }); try { - const startedAt = Date.now(); - let lastError = null; - - while (Date.now() - startedAt < timeoutMs) { - assertNoFatalLogs(logs.value); - - if (spawnError !== null) { - throw new Error(`Packaged Electron app failed to launch: ${spawnError.message}`); - } - - if (exitCode !== null || signalCode !== null) { - throw new Error( - `Packaged Electron app exited before readiness: code=${exitCode} signal=${signalCode}` - ); - } - - try { - const response = await fetchWithTimeout(smokeUrl, 1_000); - if (response.status === 200) { - assertNoFatalLogs(logs.value); - console.log(`[electron-smoke] ready: ${smokeUrl} returned HTTP 200`); - await settleAfterReady({ - getExitState: () => ({ exitCode, signalCode }), - logs, - settleMs, - }); - console.log(`[electron-smoke] stable for ${settleMs}ms after readiness`); - return; - } - lastError = new Error(`HTTP ${response.status}`); - } catch (error) { - lastError = error; - } - - await new Promise((resolve) => setTimeout(resolve, 500)); - } - - throw new Error( - `Packaged Electron app did not serve ${smokeUrl} within ${timeoutMs}ms. Last error: ${ - lastError instanceof Error ? lastError.message : String(lastError) - }` - ); + await waitForReady({ logs, smokeUrl, timeoutMs, settleMs, exitState }); + return logs.value; } catch (error) { if (!streamLogs) { printLogTail(logs.value); @@ -519,6 +547,43 @@ async function main() { } finally { await stopApp(child); await waitForPortClosed(smokeUrl); + } +} + +async function main() { + const appExecutable = discoverPackagedExecutable(); + assertExecutableExists(appExecutable); + + const smokeUrl = process.env.ELECTRON_SMOKE_URL || DEFAULT_URL; + const timeoutMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_TIMEOUT_MS, DEFAULT_TIMEOUT_MS); + const settleMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_SETTLE_MS, DEFAULT_SETTLE_MS); + const streamLogs = process.env.ELECTRON_SMOKE_STREAM_LOGS === "1"; + // #7592: rerun against the SAME (persisted) DATA_DIR and assert the second + // launch selected a native SQLite driver, not the sql.js WASM fallback. + const coldRestart = process.env.ELECTRON_SMOKE_COLD_RESTART === "1"; + const dataDir = + process.env.ELECTRON_SMOKE_DATA_DIR || + (await mkdtemp(join(tmpdir(), "omniroute-electron-smoke-"))); + const removeDataDir = + !process.env.ELECTRON_SMOKE_DATA_DIR && process.env.ELECTRON_SMOKE_KEEP_DATA !== "1"; + + try { + await launchAndCollectLogs({ appExecutable, smokeUrl, dataDir, timeoutMs, settleMs, streamLogs }); + + if (!coldRestart) return; + + console.log("[electron-smoke] cold-restart: relaunching against the same DATA_DIR"); + const secondLaunchLogs = await launchAndCollectLogs({ + appExecutable, + smokeUrl, + dataDir, + timeoutMs, + settleMs, + streamLogs, + }); + assertNativeDriverSelected(secondLaunchLogs); + console.log("[electron-smoke] cold-restart: native SQLite driver confirmed on second launch"); + } finally { if (removeDataDir) { await rm(dataDir, { recursive: true, force: true }); } diff --git a/src/app/(dashboard)/dashboard/analytics/ProviderUtilizationTab.tsx b/src/app/(dashboard)/dashboard/analytics/ProviderUtilizationTab.tsx index b8d62d96ce6..84e99583c13 100644 --- a/src/app/(dashboard)/dashboard/analytics/ProviderUtilizationTab.tsx +++ b/src/app/(dashboard)/dashboard/analytics/ProviderUtilizationTab.tsx @@ -3,6 +3,7 @@ import { useTranslations } from "next-intl"; import { useCallback, useEffect, useMemo, useState } from "react"; import { useProviderNodeMap, resolveProviderName } from "@/lib/display/useProviderNodeMap"; +import { getAccountDisplayName } from "@/lib/display/names"; import dynamic from "next/dynamic"; const ProviderCharts = dynamic(() => import("./components/ProviderCharts"), { ssr: false }); @@ -311,19 +312,39 @@ export default function ProviderUtilizationTab() { {latestPoints.map((point) => { const isLow = point.remainingPct <= 20; + // For Account Split, parse "provider:connectionId" and resolve display name + const colonIdx = point.provider.indexOf(":"); + const isConnectionKey = aggregateBy === "connection" && colonIdx !== -1; + const providerPart = isConnectionKey + ? point.provider.slice(0, colonIdx) + : point.provider; + const connectionId = isConnectionKey ? point.provider.slice(colonIdx + 1) : null; + const connMeta = connectionId ? data?.connectionMeta?.[connectionId] : null; + const cardTitle = isConnectionKey + ? getAccountDisplayName({ + id: connectionId ?? undefined, + email: connMeta?.email, + name: connMeta?.name, + displayName: connMeta?.displayName, + }) + : resolveProviderName(point.provider, nodeMap); + const cardSubtitle = isConnectionKey + ? `${providerPart} · account ${(connectionId ?? "").slice(0, 8)}…` + : t("providerUtilizationLatestSnapshot"); + return (
- +

- {resolveProviderName(point.provider, nodeMap)} + {cardTitle}

- {t("providerUtilizationLatestSnapshot")} + {cardSubtitle}

diff --git a/src/app/(dashboard)/dashboard/cli-code/components/ClineToolCard.tsx b/src/app/(dashboard)/dashboard/cli-code/components/ClineToolCard.tsx index c47ce772f2c..3a613056398 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/ClineToolCard.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/ClineToolCard.tsx @@ -223,8 +223,10 @@ export default function ClineToolCard({ const handleManualConfig = (config) => { if (config.model) setSelectedModel(config.model); - // (#523) Match apiKey string to key id if possible - if (config.apiKey && apiKeys?.length > 0) { + // (#523) Match apiKey string to key id if possible. + // apiKey may be a structured secret reference (object) rather than a + // plaintext string. Only match on strings. + if (typeof config.apiKey === "string" && config.apiKey && apiKeys?.length > 0) { const prefix = config.apiKey.slice(0, 8); const suffix = config.apiKey.slice(-4); const matchedKey = apiKeys.find( diff --git a/src/app/(dashboard)/dashboard/cli-code/components/DroidToolCard.tsx b/src/app/(dashboard)/dashboard/cli-code/components/DroidToolCard.tsx index e7eb31d7460..f3dcb2bfd4d 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/DroidToolCard.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/DroidToolCard.tsx @@ -102,7 +102,9 @@ export default function DroidToolCard({ if (existing.length > 0) { setModelList(existing.map((m) => m.model).filter(Boolean)); const first = existing[0]; - if (first?.apiKey) { + // apiKey may be a structured secret reference (object) rather than a + // plaintext string. Only match on strings. + if (typeof first?.apiKey === "string" && first.apiKey) { // (#523) Keys from /api/keys are masked. Match by prefix/suffix. const fileKeyPrefix = first.apiKey.slice(0, 8); const fileKeySuffix = first.apiKey.slice(-4); diff --git a/src/app/(dashboard)/dashboard/cli-code/components/GrokBuildToolCard.tsx b/src/app/(dashboard)/dashboard/cli-code/components/GrokBuildToolCard.tsx new file mode 100644 index 00000000000..ff849b22edb --- /dev/null +++ b/src/app/(dashboard)/dashboard/cli-code/components/GrokBuildToolCard.tsx @@ -0,0 +1,667 @@ +"use client"; + +import { useCallback, useEffect, useMemo, useState } from "react"; + +import { useTranslations } from "next-intl"; + +import CliStatusBadge from "./CliStatusBadge"; + +import { Button, Card, ManualConfigModal, ModelSelectModal } from "@/shared/components"; +import ProviderIcon from "@/shared/components/ProviderIcon"; +import type { ToolBatchStatus } from "@/shared/types/cliBatchStatus"; + +const SETTINGS_ENDPOINT = "/api/cli-tools/grok-build-settings"; +const PRESETS_KEY = "omniroute.grokBuildEndpointPresets"; +const CUSTOM_ENDPOINT = "__custom__"; +const SUBAGENTS = ["general-purpose", "explore", "plan"] as const; + +type SubagentType = (typeof SUBAGENTS)[number]; +type Message = { type: "success" | "error"; text: string } | null; +type ModelOption = { value: string; label?: string }; +type ApiKeyOption = { id: string; name?: string; key?: string }; +type EndpointOption = { id: string; label: string; url: string }; +type SavedEndpoint = { name: string; baseUrl: string }; +type Backup = { id: string; createdAt: string }; +type GrokModelStatus = { + model: string | null; + base_url: string | null; + context_window: number | null; +}; +type GrokStatus = { + installed?: boolean; + runnable?: boolean; + hasOmniRoute?: boolean; + apiKeyConfigured?: boolean; + configPath?: string; + config?: { + model?: GrokModelStatus | null; + subagentModels?: Partial>; + }; + error?: { message?: string } | string; +}; + +interface GrokBuildToolCardProps { + tool: { name: string; description?: string }; + isExpanded?: boolean; + onToggle?: () => void; + apiKeys?: ApiKeyOption[]; + activeProviders?: Array<{ + provider: string; + id?: string | number; + providerSpecificData?: unknown; + }>; + hasActiveProviders?: boolean; + availableModels?: ModelOption[]; + batchStatus?: ToolBatchStatus | null; + lastConfiguredAt?: string | null; +} + +const errorText = (body: unknown, fallback: string): string => { + if (!body || typeof body !== "object") return fallback; + const error = (body as { error?: unknown }).error; + if (typeof error === "string") return error; + if ( + error && + typeof error === "object" && + typeof (error as { message?: unknown }).message === "string" + ) { + return (error as { message: string }).message; + } + return fallback; +}; + +const ensureV1 = (value: string): string => { + const trimmed = value.trim().replace(/\/+$/, ""); + if (!trimmed) return ""; + return `${trimmed.replace(/(?:\/v1)+$/, "")}/v1`; +}; + +const readPresets = (): SavedEndpoint[] => { + try { + const value: unknown = JSON.parse(localStorage.getItem(PRESETS_KEY) ?? "[]"); + if (!Array.isArray(value)) return []; + return value.filter((item): item is SavedEndpoint => + Boolean( + item && + typeof item === "object" && + typeof item.name === "string" && + typeof item.baseUrl === "string" + ) + ); + } catch { + return []; + } +}; + +const getTunnelUrl = (body: unknown): string => { + if (!body || typeof body !== "object") return ""; + const record = body as Record; + for (const key of ["apiUrl", "publicUrl", "tunnelUrl"]) { + if (typeof record[key] === "string" && record[key]) return ensureV1(record[key]); + } + return ""; +}; + +const modelLabel = (type: SubagentType): string => + type === "general-purpose" ? "General purpose" : `${type[0].toUpperCase()}${type.slice(1)}`; + +/** Configure Grok Build model slots and endpoint access. */ +export default function GrokBuildToolCard({ + tool, + isExpanded = true, + onToggle = () => undefined, + apiKeys = [], + activeProviders = [], + hasActiveProviders = false, + availableModels = [], + batchStatus = null, + lastConfiguredAt = null, +}: GrokBuildToolCardProps) { + const t = useTranslations("cliTools"); + const [status, setStatus] = useState(null); + const [checking, setChecking] = useState(true); + const [applying, setApplying] = useState(false); + const [resetting, setResetting] = useState(false); + const [message, setMessage] = useState(null); + const [model, setModel] = useState(""); + const [subagentModels, setSubagentModels] = useState>>({}); + const [selectedKeyId, setSelectedKeyId] = useState(""); + const [endpoints, setEndpoints] = useState([]); + const [selectedEndpoint, setSelectedEndpoint] = useState(""); + const [customEndpoint, setCustomEndpoint] = useState(""); + const [modelTarget, setModelTarget] = useState<"main" | SubagentType | null>(null); + const [showManual, setShowManual] = useState(false); + const [backups, setBackups] = useState([]); + const [showBackups, setShowBackups] = useState(false); + const [restoringBackup, setRestoringBackup] = useState(null); + + useEffect(() => { + if (!selectedKeyId && apiKeys[0]?.id) setSelectedKeyId(apiKeys[0].id); + }, [apiKeys, selectedKeyId]); + + const hydrateStatus = useCallback((next: GrokStatus) => { + setStatus(next); + setModel(next.config?.model?.model ?? ""); + setSubagentModels( + Object.fromEntries( + SUBAGENTS.flatMap((type) => { + const value = next.config?.subagentModels?.[type]?.model; + return value ? [[type, value]] : []; + }) + ) + ); + }, []); + + const refreshStatus = useCallback(async () => { + setChecking(true); + try { + const response = await fetch(SETTINGS_ENDPOINT); + const body = (await response.json()) as GrokStatus; + if (!response.ok) throw new Error(errorText(body, "Failed to read Grok Build settings")); + hydrateStatus(body); + } catch (error) { + setMessage({ type: "error", text: error instanceof Error ? error.message : String(error) }); + } finally { + setChecking(false); + } + }, [hydrateStatus]); + + const refreshEndpoints = useCallback(async () => { + const requests = [ + fetch("/api/settings"), + fetch("/api/tunnels/cloudflared"), + fetch("/api/tunnels/tailscale"), + fetch("/api/tunnels/ngrok"), + ]; + const results = await Promise.allSettled(requests); + const bodies = await Promise.all( + results.map(async (result) => + result.status === "fulfilled" && result.value.ok ? result.value.json() : null + ) + ); + const settings = (bodies[0] ?? {}) as Record; + const next: EndpointOption[] = []; + if (typeof settings.apiPort === "number") { + next.push({ + id: "local", + label: "Local", + url: `http://127.0.0.1:${settings.apiPort}/v1`, + }); + } + if (typeof settings.cloudUrl === "string" && typeof settings.machineId === "string") { + next.push({ + id: "cloud", + label: "Cloud", + url: ensureV1(`${settings.cloudUrl.replace(/\/+$/, "")}/${settings.machineId}`), + }); + } + ["cloudflared", "tailscale", "ngrok"].forEach((name, index) => { + const url = getTunnelUrl(bodies[index + 1]); + if (url) next.push({ id: name, label: name, url }); + }); + readPresets().forEach((preset, index) => { + next.push({ id: `saved-${index}`, label: preset.name, url: ensureV1(preset.baseUrl) }); + }); + next.push({ id: CUSTOM_ENDPOINT, label: "Custom", url: "" }); + setEndpoints(next); + setSelectedEndpoint((current) => current || next[0]?.id || CUSTOM_ENDPOINT); + }, []); + + const refreshBackups = useCallback(async () => { + try { + const response = await fetch("/api/cli-tools/backups?tool=grok-build"); + const body = (await response.json()) as { backups?: Backup[] }; + if (response.ok) setBackups(body.backups ?? []); + } catch { + setBackups([]); + } + }, []); + + useEffect(() => { + if (!isExpanded) return; + void Promise.all([refreshStatus(), refreshEndpoints(), refreshBackups()]); + }, [isExpanded, refreshBackups, refreshEndpoints, refreshStatus]); + + const baseUrl = useMemo(() => { + if (selectedEndpoint === CUSTOM_ENDPOINT) return ensureV1(customEndpoint); + return endpoints.find((endpoint) => endpoint.id === selectedEndpoint)?.url ?? ""; + }, [customEndpoint, endpoints, selectedEndpoint]); + + const manualToml = useMemo(() => { + const contextWindowFor = (selected: string): number => + ( + availableModels.find((candidate) => candidate.value === selected) as + (ModelOption & { contextWindow?: number; contextLength?: number }) | undefined + )?.contextWindow ?? + ( + availableModels.find((candidate) => candidate.value === selected) as + (ModelOption & { contextWindow?: number; contextLength?: number }) | undefined + )?.contextLength ?? + 200000; + const mainModel = model || "provider/model-id"; + const blocks = [ + `[models]\ndefault = "omniroute"`, + `[model.omniroute]\nmodel = "${mainModel}"\nbase_url = "${baseUrl || "http://127.0.0.1:/v1"}"\nname = "OmniRoute"\ndescription = "Routed via OmniRoute gateway"\napi_backend = "chat_completions"\napi_key = ""\ncontext_window = ${contextWindowFor(mainModel)}`, + ]; + const mappings: string[] = []; + for (const type of SUBAGENTS) { + const selected = subagentModels[type]?.trim(); + if (!selected) continue; + const slot = `omniroute-${type}`; + mappings.push(`${type} = "${slot}"`); + blocks.push( + `[model.${slot}]\nmodel = "${selected}"\nbase_url = "${baseUrl || "http://127.0.0.1:/v1"}"\nname = "OmniRoute ${type}"\ndescription = "Routed via OmniRoute gateway"\napi_backend = "chat_completions"\napi_key = ""\ncontext_window = ${contextWindowFor(selected)}` + ); + } + if (mappings.length) blocks.splice(1, 0, `[subagents.models]\n${mappings.join("\n")}`); + return `${blocks.join("\n\n")}\n`; + }, [availableModels, baseUrl, model, subagentModels]); + + const apply = async () => { + setApplying(true); + setMessage(null); + try { + const desiredSubagents = Object.fromEntries( + SUBAGENTS.flatMap((type) => { + const selected = subagentModels[type]?.trim(); + return selected ? [[type, { model: selected }]] : []; + }) + ); + const selectedContext = (selected: string): number | undefined => { + const option = availableModels.find((candidate) => candidate.value === selected) as + (ModelOption & { contextWindow?: number; contextLength?: number }) | undefined; + return option?.contextWindow ?? option?.contextLength; + }; + const response = await fetch(SETTINGS_ENDPOINT, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + baseUrl, + keyId: selectedKeyId || null, + model, + contextWindow: selectedContext(model), + subagentModels: Object.fromEntries( + Object.entries(desiredSubagents).map(([type, entry]) => [ + type, + { ...entry, contextWindow: selectedContext(entry.model) }, + ]) + ), + }), + }); + const body: unknown = await response.json(); + if (!response.ok) throw new Error(errorText(body, "Failed to apply settings")); + setMessage({ type: "success", text: "Grok Build settings applied." }); + await Promise.all([refreshStatus(), refreshBackups()]); + } catch (error) { + setMessage({ type: "error", text: error instanceof Error ? error.message : String(error) }); + } finally { + setApplying(false); + } + }; + + const reset = async () => { + setResetting(true); + setMessage(null); + try { + const response = await fetch(SETTINGS_ENDPOINT, { method: "DELETE" }); + const body: unknown = await response.json(); + if (!response.ok) throw new Error(errorText(body, "Failed to reset settings")); + setMessage({ type: "success", text: "Grok Build settings reset." }); + await Promise.all([refreshStatus(), refreshBackups()]); + } catch (error) { + setMessage({ type: "error", text: error instanceof Error ? error.message : String(error) }); + } finally { + setResetting(false); + } + }; + + const selectModel = (selection: unknown) => { + const value = (selection as { value?: unknown })?.value; + if (typeof value !== "string") return; + if (modelTarget === "main") setModel(value); + else if (modelTarget) setSubagentModels((current) => ({ ...current, [modelTarget]: value })); + setModelTarget(null); + }; + + const restoreBackup = async (backupId: string) => { + setRestoringBackup(backupId); + setMessage(null); + try { + const response = await fetch("/api/cli-tools/backups", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ tool: "grok-build", backupId }), + }); + const body: unknown = await response.json(); + if (!response.ok) throw new Error(errorText(body, "Failed to restore backup")); + setMessage({ type: "success", text: "Grok Build backup restored." }); + await Promise.all([refreshStatus(), refreshBackups()]); + } catch (error) { + setMessage({ type: "error", text: error instanceof Error ? error.message : String(error) }); + } finally { + setRestoringBackup(null); + } + }; + + const configured = Boolean(status?.hasOmniRoute); + const cliReady = Boolean(status?.installed && status?.runnable); + const effectiveConfigStatus = status + ? cliReady + ? configured + ? "configured" + : "not_configured" + : "not_installed" + : (batchStatus?.config.status ?? null); + const rowClass = "flex items-center gap-2"; + const labelClass = "w-32 shrink-0 text-right text-sm font-semibold text-text-main"; + const inputClass = + "min-w-0 flex-1 rounded border border-border bg-surface px-2 py-1.5 text-xs focus:outline-none focus:ring-1 focus:ring-primary/50"; + + return ( + +
+
+
+ +
+
+
+

{tool.name}

+ +
+

{tool.description}

+
+
+ + expand_more + +
+ + {isExpanded && ( +
+ {checking && ( +
+ progress_activity + {t("checkingCli", { tool: "Grok Build" })} +
+ )} + + {!checking && status && !cliReady && ( +
+ warning +
+

+ {status.installed + ? t("cliNotRunnable", { tool: "Grok Build" }) + : t("cliNotInstalled", { tool: "Grok Build" })} +

+

+ Direct Apply needs the Grok Build CLI on this computer. Manual Config stays + available. +

+
+
+ )} + + {status?.config?.model?.base_url && ( +
+ {t("current")} + + arrow_forward + + + {status.config.model.base_url} + +
+ )} + +
+ + + arrow_forward + + +
+ {selectedEndpoint === CUSTOM_ENDPOINT && ( +
+ Custom URL + + arrow_forward + + setCustomEndpoint(event.target.value)} + placeholder="https://gateway.example/v1" + /> +
+ )} + +
+ + + arrow_forward + + +
+ +
+ {t("model")} + + arrow_forward + + + setModel(event.target.value)} + placeholder={availableModels[0]?.value || "provider/model-id"} + /> + {model && ( + + )} +
+ +
+
+ Subagent model overrides +
+ {SUBAGENTS.map((type) => ( +
+ + {modelLabel(type)} + + + arrow_forward + + + + setSubagentModels((current) => ({ ...current, [type]: event.target.value })) + } + placeholder={`Use ${model || "the main model"}`} + /> + {subagentModels[type] && ( + + )} +
+ ))} + + {message && ( +

+ {message.text} +

+ )} + +
+ + + +
+ +
+ + {showBackups && ( +
+

+ history + {t("configBackups")} +

+ {backups.length === 0 ? ( +

{t("noBackupsYet")}

+ ) : ( +
+ {backups.map((backup) => ( +
+ + description + + + {backup.id} + + + {new Date(backup.createdAt).toLocaleString()} + + +
+ ))} +
+ )} +
+ )} +
+ )} + + {modelTarget && ( + setModelTarget(null)} + onSelect={selectModel} + selectedModel={modelTarget === "main" ? model : (subagentModels[modelTarget] ?? "")} + activeProviders={activeProviders} + title={`Select ${modelTarget === "main" ? "main" : modelLabel(modelTarget)} model`} + /> + )} + setShowManual(false)} + title="Grok Build manual config" + configs={[{ filename: "~/.grok/config.toml", content: manualToml }]} + /> + + ); +} diff --git a/src/app/(dashboard)/dashboard/cli-code/components/KiloToolCard.tsx b/src/app/(dashboard)/dashboard/cli-code/components/KiloToolCard.tsx index 9b597e482c6..d682f913ca1 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/KiloToolCard.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/KiloToolCard.tsx @@ -209,8 +209,10 @@ export default function KiloToolCard({ const handleManualConfig = (config) => { if (config.model) setSelectedModel(config.model); - // (#523) Match apiKey string to key id if possible - if (config.apiKey && apiKeys?.length > 0) { + // (#523) Match apiKey string to key id if possible. + // apiKey may be a structured secret reference (object) rather than a + // plaintext string. Only match on strings. + if (typeof config.apiKey === "string" && config.apiKey && apiKeys?.length > 0) { const prefix = config.apiKey.slice(0, 8); const suffix = config.apiKey.slice(-4); const matchedKey = apiKeys.find( diff --git a/src/app/(dashboard)/dashboard/cli-code/components/OpenClawToolCard.tsx b/src/app/(dashboard)/dashboard/cli-code/components/OpenClawToolCard.tsx index cf20a9b8f09..6c904958b3c 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/OpenClawToolCard.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/OpenClawToolCard.tsx @@ -94,7 +94,9 @@ export default function OpenClawToolCard({ } // (#523) Keys from /api/keys are masked (first 8 + "****" + last 4). // Match by prefix/suffix instead of exact comparison. - if (provider.apiKey) { + // apiKey may be a structured secret reference (object) rather than a + // plaintext string, e.g. OpenClaw SecretRefs. Only match on strings. + if (typeof provider.apiKey === "string" && provider.apiKey) { const fileKeyPrefix = provider.apiKey.slice(0, 8); const fileKeySuffix = provider.apiKey.slice(-4); const matchedKey = apiKeys?.find( diff --git a/src/app/(dashboard)/dashboard/cli-code/components/ToolDetailClient.tsx b/src/app/(dashboard)/dashboard/cli-code/components/ToolDetailClient.tsx index 96771fd0c6d..fdad0931f86 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/ToolDetailClient.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/ToolDetailClient.tsx @@ -15,6 +15,7 @@ import { DefaultToolCard, DroidToolCard, HermesAgentToolCard, + GrokBuildToolCard, KiloToolCard, OpenClawToolCard, } from "./index"; @@ -268,6 +269,8 @@ export default function ToolDetailClient({ toolId, category }: ToolDetailClientP return ; case "hermes-agent": return ; + case "grok-build": + return ; case "antigravity": return ; case "custom": diff --git a/src/app/(dashboard)/dashboard/cli-code/components/index.tsx b/src/app/(dashboard)/dashboard/cli-code/components/index.tsx index bae6f2571b4..a4dd2228475 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/index.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/index.tsx @@ -9,3 +9,4 @@ export { default as AntigravityToolCard } from "./AntigravityToolCard"; export { default as CopilotToolCard } from "./CopilotToolCard"; export { default as CustomCliCard } from "./CustomCliCard"; export { default as HermesAgentToolCard } from "./HermesAgentToolCard"; +export { default as GrokBuildToolCard } from "./GrokBuildToolCard"; diff --git a/src/app/(dashboard)/dashboard/combos/AutoComboCatalog.tsx b/src/app/(dashboard)/dashboard/combos/AutoComboCatalog.tsx index a072fa8f1b3..c3d69551487 100644 --- a/src/app/(dashboard)/dashboard/combos/AutoComboCatalog.tsx +++ b/src/app/(dashboard)/dashboard/combos/AutoComboCatalog.tsx @@ -1,19 +1,66 @@ "use client"; -import { useState } from "react"; +import { useState, useCallback } from "react"; import { useTranslations } from "next-intl"; import { Card } from "@/shared/components"; -import { AUTO_COMBO_TEMPLATES } from "@/domain/assessment/types"; +import { AUTO_COMBO_TEMPLATES, type AutoComboTemplate } from "@/domain/assessment/types"; // Informational catalog of zero-config auto-routing combos. // Auto combos are resolved at request time by the chat handler based on the // currently connected providers / models — they have no row in the combos // table, so they were previously invisible in the UI. This panel surfaces // the static catalog (name, intent, categories, tiers, strategy) so users -// can discover the auto/ prefix without reading source. -export default function AutoComboCatalog() { +// can discover the auto/ prefix without reading source. Duplicate icon lets +// you materialize a snapshot into an editable static combo you can customize. +export default function AutoComboCatalog({ + onComboCreated, +}: { + onComboCreated?: (comboId: string) => void; +}) { const t = useTranslations("combos"); const [open, setOpen] = useState(false); + const [duplicatingName, setDuplicatingName] = useState(null); + + const handleDuplicateTemplate = useCallback( + async (template: AutoComboTemplate) => { + if ( + !confirm( + `${t("duplicateAutoComboConfirm", { name: template.name })}\n\n${t("duplicateAutoComboSnapshotMsg")}` + ) + ) + return; + + setDuplicatingName(template.name); + try { + const res = await fetch("/api/combos/duplicate", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ name: template.name, strategy: template.strategy }), + }); + + if (!res.ok) { + const data = await res.json().catch(() => ({})); + alert( + `${t("duplicateAutoComboFailedPrefix")} ${data.error || t("duplicateAutoComboUnknownError")}` + ); + return; + } + + const combo = await res.json(); + + // Notify parent page to re-fetch combos so the new card renders. + onComboCreated?.(String(combo.id)); + } catch (err) { + console.error("Error duplicating auto-combo:", err); + alert( + `${t("duplicateAutoComboFailedPrefix")} ${err instanceof Error ? err.message : t("duplicateAutoComboUnknownError")}` + ); + } finally { + setDuplicatingName(null); + } + }, + [t, onComboCreated] + ); return ( @@ -44,8 +91,21 @@ export default function AutoComboCatalog() { {AUTO_COMBO_TEMPLATES.map((tpl) => (
+ +
{tpl.name} diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index fd9d1e3b7e2..fceada4eb56 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -879,6 +879,15 @@ export default function CombosPage() { } }; + const handleComboCreated = async (comboId: string) => { + await fetchData(); + // Wait for React to re-render the new card, then scroll it into view. + setTimeout(() => { + const el = document.querySelector(`[data-testid="combo-card-${comboId}"]`); + if (el) el.scrollIntoView({ behavior: "auto", block: "center" }); + }, 0); + }; + const handleDelete = async (id) => { if (!confirm(t("deleteConfirm"))) return; try { @@ -1104,7 +1113,7 @@ export default function CombosPage() {
- + {t("skipPassword")} + {skipSecurity && ( +

+ {t("securityDescSkipWarning")} +

+ )} {!skipSecurity && (

{t("providerDesc")}

- -
- - {t("freeProviders.orUseApiKey")} - -
-
- {COMMON_PROVIDERS.map((p) => ( - - ))} -
- {selectedProvider && ( + {skipSecurity && ( +
+

{t("providerRequiresPassword")}

+
+ )} + {!skipSecurity && } + {!skipSecurity && ( +
+ + {t("freeProviders.orUseApiKey")} + +
+ )} + {!skipSecurity && ( +
+ {COMMON_PROVIDERS.map((p) => ( + + ))} +
+ )} + {!skipSecurity && selectedProvider && (
)} - {currentStep.id === "provider" && ( + {currentStep.id === "provider" && !skipSecurity ? ( - )} + ) : null} {currentStep.id === "test" && (
) : null; - if (hasPassword === null || setupComplete === null || oidcEnabled === null) { + if ( + hasPassword === null || + setupComplete === null || + oidcEnabled === null || + (oidcEnabled && oidcDisablePasswordLogin === null) + ) { return (
{nodeWarningBanner} @@ -235,60 +244,84 @@ export default function LoginPage() {

{t("signIn")}

-

{t("enterPassword")}

+

+ {oidcEnabled && oidcDisablePasswordLogin + ? t("continueWithOidc") + : t("enterPassword")} +

-
-
- - setPassword(e.target.value)} - required - autoFocus - className="h-11" - /> - {error && ( -

- error - {error} -

- )} -

{t("defaultPasswordHint")}

-
- - - - {oidcEnabled && ( -
+ {oidcEnabled && oidcDisablePasswordLogin ? ( +
+ ) : ( + <> +
+
+ + setPassword(e.target.value)} + required + autoFocus + className="h-11" + /> + {error && ( +

+ error + {error} +

+ )} +

{t("defaultPasswordHint")}

+
+ + + + + {oidcEnabled && ( +
+ +
+ )} + )} - + {!oidcEnabled && ( + + )}
diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index 85bb06ddb8e..c7eeeb37542 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -3722,7 +3722,12 @@ "errorDescription": "لم نتمكن من تحميل بيانات المجموعة في الوقت الحالي. تحقق من اتصالك وحاول مرة أخرى.", "errorId": "معرّف الخطأ: {id}", "errorRetry": "حاول مرة أخرى", - "comboLabel": "كومبو" + "comboLabel": "كومبو", + "duplicateAutoComboConfirm": "إنشاء مجموعة ثابتة من \"{name}\"؟", + "duplicateAutoComboSnapshotMsg": "سيؤدي هذا إلى التقاط المزودين/النماذج المتصلة حاليًا التي تطابق هذا القالب في مجموعة قابلة للتحرير.", + "duplicateAutoComboFailedPrefix": "فشل تكرار المجموعة التلقائية:", + "duplicateAutoComboUnknownError": "خطأ غير معروف", + "duplicateAutoComboTitle": "إنشاء مجموعة ثابتة من {name}" }, "costs": { "title": "التكاليف", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 6bf2e21e550..605fe9ed184 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -3722,7 +3722,12 @@ "errorDescription": "Hazırda kombinasiyalı məlumatları yükləyə bilmirik. Bağlantınızı yoxlayın və yenidən cəhd edin.", "errorId": "Xəta ID: {id}", "errorRetry": "Yenidən Cəhd Et", - "comboLabel": "Kombinasiya" + "comboLabel": "Kombinasiya", + "duplicateAutoComboConfirm": "\"{name}\"-dan statik kombo yaratmaq?", + "duplicateAutoComboSnapshotMsg": "Bu, bu şablona uyğun olan hazırkı qoşulmuş provayderləri/modeləri redaktə oluna bilən kombo kimi saxlayacaq.", + "duplicateAutoComboFailedPrefix": "Avtokombo kopyalanması uğursuz oldu:", + "duplicateAutoComboUnknownError": "Naməlum xəta", + "duplicateAutoComboTitle": "{name}-dan statik kombo yarat" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index a8b6e486b3f..b2dfd1176a0 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -3722,7 +3722,12 @@ "errorDescription": "Не успяхме да заредим данните за комбинирането в момента. Проверете връзката си и опитайте отново.", "errorId": "Идентификатор на грешка: {id}", "errorRetry": "Опитай отново", - "comboLabel": "Комбо" + "comboLabel": "Комбо", + "duplicateAutoComboConfirm": "Да създадете статично комбо от \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Това ще направи моментна снимка на текущо свързаните доставчици/модели, които съвпадат с този шаблон, в редактируемо комбо.", + "duplicateAutoComboFailedPrefix": "Неуспешно копиране на автоматично комбо:", + "duplicateAutoComboUnknownError": "Неизвестна грешка", + "duplicateAutoComboTitle": "Създайте статично комбо от {name}" }, "costs": { "title": "Разходи", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index aadc026d06a..34ecf02c74f 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -3722,7 +3722,12 @@ "errorDescription": "আমরা এখন কম্বো ডেটা লোড করতে পারিনি। আপনার সংযোগ পরীক্ষা করুন এবং আবার চেষ্টা করুন।", "errorId": "ত্রুটি আইডি: {id}", "errorRetry": "পুনরায় চেষ্টা করুন", - "comboLabel": "কম্বো" + "comboLabel": "কম্বো", + "duplicateAutoComboConfirm": "\"{name}\" থেকে একটি স্ট্যাটিক কম্বো তৈরি করবেন?", + "duplicateAutoComboSnapshotMsg": "এটি এই টেমপ্লেটের সাথে মিলে যায় এমন বর্তমান সংযুক্ত প্রদানকারী/মডেলগুলিকে একটি সম্পাদনযোগ্য কম্বোতে স্ন্যাপশট নেবে।", + "duplicateAutoComboFailedPrefix": "অটোকম্বো ডুপ্লিকেশন ব্যর্থ:", + "duplicateAutoComboUnknownError": "অজানা ত্রুটি", + "duplicateAutoComboTitle": "{name} থেকে একটি স্ট্যাটিক কম্বো তৈরি করুন" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 5e95c8f3573..18761a6f44f 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -3722,7 +3722,12 @@ "errorDescription": "Nyní se nám nepodařilo načíst data pro kombinaci. Zkontrolujte své připojení a zkuste to znovu.", "errorId": "Chyba ID: {id}", "errorRetry": "Zkusit znovu", - "comboLabel": "Kombinace" + "comboLabel": "Kombinace", + "duplicateAutoComboConfirm": "Vytvořit statickou kombinaci z \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Tím se zachytí aktuálně připojení poskytovatelé/modely, které odpovídají této šabloně, do upravitelné kombinace.", + "duplicateAutoComboFailedPrefix": "Duplikace automatické kombinace selhala:", + "duplicateAutoComboUnknownError": "Neznámá chyba", + "duplicateAutoComboTitle": "Vytvořte statickou kombinaci z {name}" }, "costs": { "title": "Náklady", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index fdc9a5fc894..cd127ca8a33 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -3722,7 +3722,12 @@ "errorDescription": "Vi kunne ikke indlæse kombinationsdata lige nu. Tjek din forbindelse og prøv igen.", "errorId": "Fejl ID: {id}", "errorRetry": "Prøv igen", - "comboLabel": "Kombination" + "comboLabel": "Kombination", + "duplicateAutoComboConfirm": "Opret en statisk kombination fra \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Dette vil tage et øjebliksbillede af de aktuelt tilsluttede udbydere/modeller, der matcher denne skabelon, i en redigerbar kombination.", + "duplicateAutoComboFailedPrefix": "Kopiering af auto-kombination mislykkedes:", + "duplicateAutoComboUnknownError": "Ukendt fejl", + "duplicateAutoComboTitle": "Opret en statisk kombination fra {name}" }, "costs": { "title": "Omkostninger", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index 2b4c363146f..d1f60bb1425 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -3722,7 +3722,12 @@ "errorDescription": "Wir konnten die Kombinationsdaten momentan nicht laden. Überprüfen Sie Ihre Verbindung und versuchen Sie es erneut.", "errorId": "Fehler-ID: {id}", "errorRetry": "Versuche es erneut", - "comboLabel": "Kombination" + "comboLabel": "Kombination", + "duplicateAutoComboConfirm": "Eine statische Kombination aus \"{name}\" erstellen?", + "duplicateAutoComboSnapshotMsg": "Dadurch werden die aktuell verbundenen Anbieter/Modelle, die dieser Vorlage entsprechen, in einer bearbeitbaren Kombination gespeichert.", + "duplicateAutoComboFailedPrefix": "Automatische Kombination konnte nicht dupliziert werden:", + "duplicateAutoComboUnknownError": "Unbekannter Fehler", + "duplicateAutoComboTitle": "Erstelle eine statische Kombination aus {name}" }, "costs": { "title": "Kosten", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 92b0e725b43..1c176ff464c 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -3727,7 +3727,12 @@ "errorDescription": "We could not load combo data right now. Check your connection and try again.", "errorId": "Error ID: {id}", "errorRetry": "Try Again", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Create a static combo from \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "This will snapshot the currently connected providers/models that match this template into an editable combo.", + "duplicateAutoComboFailedPrefix": "Failed to duplicate auto-combo:", + "duplicateAutoComboUnknownError": "Unknown error", + "duplicateAutoComboTitle": "Create a static combo from {name}" }, "costs": { "title": "Costs", @@ -4917,7 +4922,9 @@ "multiProvider": "Multi-Provider", "usageTracking": "Usage Tracking", "securityDesc": "Set a password to protect your dashboard, or skip for now.", + "securityDescSkipWarning": "⚠️ Without a password, you won't be able to add providers during setup. You can add them later from the dashboard after setting a password.", "providerDesc": "Connect your first AI provider. You can add more later.", + "providerRequiresPassword": "You need to set a password first to add providers. Please go back to the security step and set a password, or add providers later from the dashboard.", "apiKeyRequired": "API Key (required)", "customUrlOptional": "Custom URL (optional)", "testDesc": "Let's verify your provider connection works.", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index a6b357e2229..94b17341c19 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -3722,7 +3722,12 @@ "errorDescription": "No pudimos cargar los datos del combo en este momento. Verifica tu conexión y vuelve a intentarlo.", "errorId": "Error ID: {id}", "errorRetry": "Inténtalo de nuevo", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "¿Crear una combinación estática de \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Esto capturará los proveedores/modelos conectados actualmente que coincidan con esta plantilla en una combinación editable.", + "duplicateAutoComboFailedPrefix": "Error al duplicar la combinación automática:", + "duplicateAutoComboUnknownError": "Error desconocido", + "duplicateAutoComboTitle": "Crear una combinación estática de {name}" }, "costs": { "title": "Costos", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 35f12114efb..851f93aba97 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -3722,7 +3722,12 @@ "errorDescription": "در حال حاضر نمی‌توانیم داده‌های ترکیبی را بارگذاری کنیم. اتصال خود را بررسی کنید و دوباره تلاش کنید.", "errorId": "شناسه خطا: {id}", "errorRetry": "دوباره تلاش کنید", - "comboLabel": "ترکیب" + "comboLabel": "ترکیب", + "duplicateAutoComboConfirm": "ایجاد یک ترکیب ثابت از \"{name}\"؟", + "duplicateAutoComboSnapshotMsg": "این ارائه‌دهندگان/مدل‌های متصل فعلی که با این قالب مطابقت دارند را در یک ترکیب قابل ویرایش ذخیره می‌کند.", + "duplicateAutoComboFailedPrefix": "تکرار ترکیب خودکار ناموفق بود:", + "duplicateAutoComboUnknownError": "خطای ناشناخته", + "duplicateAutoComboTitle": "ایجاد یک ترکیب ثابت از {name}" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index fd15e9dfaa0..9149e2ec4fd 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -3722,7 +3722,12 @@ "errorDescription": "Emme voi ladata yhdistelmädataa juuri nyt. Tarkista yhteytesi ja yritä uudelleen.", "errorId": "Virhe ID: {id}", "errorRetry": "Yritä uudelleen", - "comboLabel": "Yhdistelmä" + "comboLabel": "Yhdistelmä", + "duplicateAutoComboConfirm": "Luodaanko staattinen yhdistelmä \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Tämä tallentaa tämän mallin mukaiset tällä hetkellä yhdistetyt tarjoajat/mallit muokattavaan yhdistelmään.", + "duplicateAutoComboFailedPrefix": "Automaattisen yhdistelmän kaksoiskappaleen luonti epäonnistui:", + "duplicateAutoComboUnknownError": "Tuntematon virhe", + "duplicateAutoComboTitle": "Luo staattinen yhdistelmä kohteesta {name}" }, "costs": { "title": "Kustannukset", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 64fc6996743..72a3792df2b 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -3722,7 +3722,12 @@ "errorDescription": "Les données des combos ne peuvent pas être chargées pour le moment. Vérifiez votre connexion et réessayez.", "errorId": "ID d'erreur : {id}", "errorRetry": "Réessayer", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Créer une combinaison statique à partir de \"{name}\" ?", + "duplicateAutoComboSnapshotMsg": "Cela capturera les fournisseurs/modèles actuellement connectés qui correspondent à ce modèle dans une combinaison modifiable.", + "duplicateAutoComboFailedPrefix": "Échec de la duplication de la combinaison automatique :", + "duplicateAutoComboUnknownError": "Erreur inconnue", + "duplicateAutoComboTitle": "Créer une combinaison statique à partir de {name}" }, "costs": { "title": "Coûts", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 5e21f097c18..a1a38f1980f 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -3722,7 +3722,12 @@ "errorDescription": "અમે હાલમાં કોમ્બો ડેટા લોડ કરી શક્યા નથી. તમારી કનેક્શન તપાસો અને ફરી પ્રયાસ કરો.", "errorId": "ભૂલ આઈડી: {id}", "errorRetry": "ફરીથી પ્રયાસ કરો", - "comboLabel": "કોમ્બો" + "comboLabel": "કોમ્બો", + "duplicateAutoComboConfirm": "\"{name}\" માંથી સ્ટેટિક કોમ્બો બનાવશો?", + "duplicateAutoComboSnapshotMsg": "આ ટેમપ્લેટ સાથે મેચ થતા વર્તમાન જોડાયેલા પ્રદાતાઓ/મોડેલને સંપાદનીય કોમ્બોમાં સ્નેપશોટ લેશે.", + "duplicateAutoComboFailedPrefix": "ઓટોકોમ્બો ડુપ્લિકેટ કરવામાં નિષ્ફળ:", + "duplicateAutoComboUnknownError": "અજ્ઞાત ભૂલ", + "duplicateAutoComboTitle": "{name} માંથી સ્ટેટિક કોમ્બો બનાવો" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 90db68373ac..96007496ae6 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -3722,7 +3722,12 @@ "errorDescription": "לא הצלחנו לטעון את נתוני הקומבו כרגע. בדוק את החיבור שלך ונסה שוב.", "errorId": "שגיאת מזהה: {id}", "errorRetry": "נסה שוב", - "comboLabel": "קומבו" + "comboLabel": "קומבו", + "duplicateAutoComboConfirm": "ליצור קומבו סטטי מ־\"{name}\"?", + "duplicateAutoComboSnapshotMsg": "פעולה זו תצלם את הספקים/מודלים המחוברים כעת התואמים לתבנית הזו לקומבו ניתן לעריכה.", + "duplicateAutoComboFailedPrefix": "כשל בהעתיק קומבו אוטומטי:", + "duplicateAutoComboUnknownError": "שגיאה לא ידועה", + "duplicateAutoComboTitle": "צור קומבו סטטי מ־{name}" }, "costs": { "title": "עלויות", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 9e353a269b1..3cbff8e2021 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -3722,7 +3722,12 @@ "errorDescription": "हम अभी कॉम्बो डेटा लोड नहीं कर सके। कृपया अपनी कनेक्शन की जांच करें और फिर से प्रयास करें।", "errorId": "त्रुटि आईडी: {id}", "errorRetry": "फिर से प्रयास करें", - "comboLabel": "कॉम्बो" + "comboLabel": "कॉम्बो", + "duplicateAutoComboConfirm": "\"{name}\" से एक स्थैतिक कॉम्बो बनाएं?", + "duplicateAutoComboSnapshotMsg": "यह इस टेम्पलेट से मेल खाने वाले वर्तमान जुड़े प्रदाताओं/मॉडल को संपादन योग्य कॉम्बो में स्नैपशॉट लेगा।", + "duplicateAutoComboFailedPrefix": "ऑटोकॉम्बो डुप्लिकेट करने में विफल:", + "duplicateAutoComboUnknownError": "अज्ञात त्रुटि", + "duplicateAutoComboTitle": "{name} से एक स्थैतिक कॉम्बो बनाएं" }, "costs": { "title": "लागत", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 4f8ee2278b9..6f723ed8531 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -3722,7 +3722,12 @@ "errorDescription": "Jelenleg nem tudtuk betölteni a kombinált adatokat. Ellenőrizze a kapcsolatát, és próbálja újra.", "errorId": "Hibaazonosító: {id}", "errorRetry": "Próbáld újra", - "comboLabel": "Kombó" + "comboLabel": "Kombó", + "duplicateAutoComboConfirm": "Létrehoz egy statikus kombinációt a(z) \"{name}\"-ból?", + "duplicateAutoComboSnapshotMsg": "Ez rögzíti az éppen csatlakoztatott szolgáltatókat/modelleket, amelyek megfelelnek ennek a sablonnak, egy szerkeszthető kombinációba.", + "duplicateAutoComboFailedPrefix": "Az automatikus kombináció duplikálása sikertelen:", + "duplicateAutoComboUnknownError": "Ismeretlen hiba", + "duplicateAutoComboTitle": "Hozzon létre statikus kombinációt a(z) {name}-ból" }, "costs": { "title": "Költségek", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index bc841a35d6a..f6aa07123ae 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -3722,7 +3722,12 @@ "errorDescription": "Kami tidak dapat memuat data kombinasi saat ini. Periksa koneksi Anda dan coba lagi.", "errorId": "Error ID: {id}", "errorRetry": "Coba Lagi", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Buat kombo statis dari \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Ini akan mengambil snapshot penyedia/model yang terhubung saat ini yang cocok dengan templat ini ke dalam kombo yang dapat diedit.", + "duplicateAutoComboFailedPrefix": "Gagal menduplikasi kombo otomatis:", + "duplicateAutoComboUnknownError": "Kesalahan tidak diketahui", + "duplicateAutoComboTitle": "Buat kombo statis dari {name}" }, "costs": { "title": "Biaya", diff --git a/src/i18n/messages/in.json b/src/i18n/messages/in.json index b9c7e6706c7..5f2587eb6a8 100644 --- a/src/i18n/messages/in.json +++ b/src/i18n/messages/in.json @@ -3722,7 +3722,12 @@ "errorDescription": "Kami tidak dapat memuat data kombinasi saat ini. Periksa koneksi Anda dan coba lagi.", "errorId": "ID Kesalahan: {id}", "errorRetry": "Coba Lagi", - "comboLabel": "Kombinasi" + "comboLabel": "Kombinasi", + "duplicateAutoComboConfirm": "Buat kombo statis dari \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Ini akan mengambil snapshot penyedia/model yang terhubung saat ini yang cocok dengan templat ini ke dalam kombo yang dapat diedit.", + "duplicateAutoComboFailedPrefix": "Gagal menduplikasi kombo otomatis:", + "duplicateAutoComboUnknownError": "Kesalahan tidak diketahui", + "duplicateAutoComboTitle": "Buat kombo statis dari {name}" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index d777c72bf12..902ed0795ad 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -3722,7 +3722,12 @@ "errorDescription": "Non siamo riusciti a caricare i dati del combo in questo momento. Controlla la tua connessione e riprova.", "errorId": "ID Errore: {id}", "errorRetry": "Riprova", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Creare una combinazione statica da \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Questo catturerà i fornitori/modelli attualmente connessi che corrispondono a questo modello in una combinazione modificabile.", + "duplicateAutoComboFailedPrefix": "Duplicazione della combinazione automatica fallita:", + "duplicateAutoComboUnknownError": "Errore sconosciuto", + "duplicateAutoComboTitle": "Crea una combinazione statica da {name}" }, "costs": { "title": "Costi", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index ee7b56d0743..d586218bc86 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -3722,7 +3722,12 @@ "errorDescription": "現在、コンボデータを読み込むことができません。接続を確認して、再試行してください。", "errorId": "エラー ID: {id}", "errorRetry": "もう一度試してください", - "comboLabel": "コンボ" + "comboLabel": "コンボ", + "duplicateAutoComboConfirm": "\"{name}\"から静的コンボを作成しますか?", + "duplicateAutoComboSnapshotMsg": "このテンプレートに一致する現在接続されているプロバイダー/モデルを編集可能なコンボとしてスナップショットします。", + "duplicateAutoComboFailedPrefix": "オートコンボの複製に失敗しました:", + "duplicateAutoComboUnknownError": "不明なエラー", + "duplicateAutoComboTitle": "{name}から静的コンボを作成" }, "costs": { "title": "コスト", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 3a5feebf4ef..4f10276f71b 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -3722,7 +3722,12 @@ "errorDescription": "현재 콤보 데이터를 불러올 수 없습니다. 연결을 확인하고 다시 시도하세요.", "errorId": "오류 ID: {id}", "errorRetry": "다시 시도해 주세요", - "comboLabel": "콤보" + "comboLabel": "콤보", + "duplicateAutoComboConfirm": "\"{name}\"에서 정적 콤보를 만드시겠습니까?", + "duplicateAutoComboSnapshotMsg": "이 템플릿과 일치하는 현재 연결된 제공자/모델을 편집 가능한 콤보로 스냅샷합니다.", + "duplicateAutoComboFailedPrefix": "자동 콤보 복제 실패:", + "duplicateAutoComboUnknownError": "알 수 없는 오류", + "duplicateAutoComboTitle": "{name}에서 정적 콤보 만들기" }, "costs": { "title": "비용", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 63afee6a7f7..86437f4f3ac 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -3722,7 +3722,12 @@ "errorDescription": "आम्ही सध्या कॉम्बो डेटा लोड करू शकत नाही. तुमचा कनेक्शन तपासा आणि पुन्हा प्रयत्न करा.", "errorId": "त्रुटी आयडी: {id}", "errorRetry": "पुन्हा प्रयत्न करा", - "comboLabel": "कॉम्बो" + "comboLabel": "कॉम्बो", + "duplicateAutoComboConfirm": "\"{name}\" मधून स्थिर कॉम्बो तयार करायचा?", + "duplicateAutoComboSnapshotMsg": "या टेम्पलेटशी जुळणारे सध्या कनेक्ट केलेले प्रदाता/मॉडेल्स संपादनयोग्य कॉम्बोमध्ये स्नॅपशॉट घेईल.", + "duplicateAutoComboFailedPrefix": "ऑटोकॉम्बो डुप्लिकेट करण्यात अपयश:", + "duplicateAutoComboUnknownError": "अज्ञात त्रुटी", + "duplicateAutoComboTitle": "{name} मधून स्थिर कॉम्बो तयार करा" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index b6c322c3e27..498af0d138c 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -3722,7 +3722,12 @@ "errorDescription": "Kami tidak dapat memuatkan data combo buat masa ini. Semak sambungan anda dan cuba lagi.", "errorId": "Ralat ID: {id}", "errorRetry": "Cuba Lagi", - "comboLabel": "Gabungan" + "comboLabel": "Gabungan", + "duplicateAutoComboConfirm": "Cipta kombo statik dari \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Ini akan mengambil snapshot penyedia/model yang disambungkan semasa yang sepadan dengan templat ini ke dalam kombo yang boleh diedit.", + "duplicateAutoComboFailedPrefix": "Gagal menduplikasi kombo automatik:", + "duplicateAutoComboUnknownError": "Ralat tidak diketahui", + "duplicateAutoComboTitle": "Cipta kombo statik dari {name}" }, "costs": { "title": "Kos", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 397c2068884..6422dad14d2 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -3722,7 +3722,12 @@ "errorDescription": "We konden de combogegevens op dit moment niet laden. Controleer je verbinding en probeer het opnieuw.", "errorId": "Fout-ID: {id}", "errorRetry": "Probeer het opnieuw", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Een statische combinatie maken van \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Dit maakt een momentopname van de momenteel verbonden providers/modellen die overeenkomen met dit sjabloon in een bewerkbare combinatie.", + "duplicateAutoComboFailedPrefix": "Automatische combinatie dupliceren mislukt:", + "duplicateAutoComboUnknownError": "Onbekende fout", + "duplicateAutoComboTitle": "Maak een statische combinatie van {name}" }, "costs": { "title": "Kosten", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 502bbae12e7..6f105c14c34 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -3722,7 +3722,12 @@ "errorDescription": "Vi kunne ikke laste inn kombinasjonsdata akkurat nå. Sjekk tilkoblingen din og prøv igjen.", "errorId": "Feil-ID: {id}", "errorRetry": "Prøv igjen", - "comboLabel": "Kombinasjon" + "comboLabel": "Kombinasjon", + "duplicateAutoComboConfirm": "Opprett en statisk kombinasjon fra \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Dette vil ta et øyeblikksbilde av de nåvørende tilkoblede leverandørene/modellene som matcher denne malen i en redigerbar kombinasjon.", + "duplicateAutoComboFailedPrefix": "Duplisering av auto-kombinasjon mislyktes:", + "duplicateAutoComboUnknownError": "Ukjent feil", + "duplicateAutoComboTitle": "Opprett en statisk kombinasjon fra {name}" }, "costs": { "title": "Kostnader", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 5285b3d18c3..f9dc8a149b2 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -3722,7 +3722,12 @@ "errorDescription": "Hindi namin ma-load ang combo data sa ngayon. Suriin ang iyong koneksyon at subukan muli.", "errorId": "Error ID: {id}", "errorRetry": "Subukan Muli", - "comboLabel": "Kumbinasyon" + "comboLabel": "Kumbinasyon", + "duplicateAutoComboConfirm": "Gumawa ng static combo mula sa \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Ito ay kukuha ng snapshot ng kasalukuyang nakakonektang mga provider/model na tugma sa template na ito sa isang editable na combo.", + "duplicateAutoComboFailedPrefix": "Nabigo ang pag-duplicate ng auto-combo:", + "duplicateAutoComboUnknownError": "Hindi alam na error", + "duplicateAutoComboTitle": "Gumawa ng static combo mula sa {name}" }, "costs": { "title": "Mga gastos", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index e29e73670eb..2bca5453b12 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -3722,7 +3722,12 @@ "errorDescription": "Nie mogliśmy teraz załadować danych combo. Sprawdź swoje połączenie i spróbuj ponownie.", "errorId": "Identyfikator błędu: {id}", "errorRetry": "Spróbuj ponownie", - "comboLabel": "Kombinacja" + "comboLabel": "Kombinacja", + "duplicateAutoComboConfirm": "Utworzyć statyczną kombinację z \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Spowoduje to przechwycenie aktualnie połączonych dostawców/modeli pasujących do tego szablonu w edytowalnej kombinacji.", + "duplicateAutoComboFailedPrefix": "Nie udało się skopiować automatycznej kombinacji:", + "duplicateAutoComboUnknownError": "Nieznany błąd", + "duplicateAutoComboTitle": "Utwórz statyczną kombinację z {name}" }, "costs": { "title": "Koszty", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 4ba29762d00..9a3b93def69 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -3727,7 +3727,12 @@ "errorDescription": "Não conseguimos carregar os dados do combo no momento. Verifique sua conexão e tente novamente.", "errorId": "ID de Erro: {id}", "errorRetry": "Tente Novamente", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Criar uma combinação estática de \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Isso irá capturar os provedores/modelos atualmente conectados que correspondem a este modelo em uma combinação editável.", + "duplicateAutoComboFailedPrefix": "Falha ao duplicar a combinação automática:", + "duplicateAutoComboUnknownError": "Erro desconhecido", + "duplicateAutoComboTitle": "Criar uma combinação estática de {name}" }, "costs": { "title": "Custos", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 05441018a31..dc1a90c803c 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -3722,7 +3722,12 @@ "errorDescription": "Não conseguimos carregar os dados do combo neste momento. Verifique a sua ligação e tente novamente.", "errorId": "ID de Erro: {id}", "errorRetry": "Tente Novamente", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Criar uma combinação estática de \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Isto irá capturar os provedores/modelos atualmente conectados que correspondem a este modelo em uma combinação editável.", + "duplicateAutoComboFailedPrefix": "Falha ao duplicar a combinação automática:", + "duplicateAutoComboUnknownError": "Erro desconhecido", + "duplicateAutoComboTitle": "Criar uma combinação estática de {name}" }, "costs": { "title": "Custos", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index b8ef63cbb80..4265215f2ee 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -3722,7 +3722,12 @@ "errorDescription": "Nu am putut încărca datele combo în acest moment. Verifică-ți conexiunea și încearcă din nou.", "errorId": "ID eroare: {id}", "errorRetry": "Încearcă din nou", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Creați o combinație statică din \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Acesta va captura furnizorii/modelurile conectate în prezent care se potrivesc cu acest șablon într-o combinație editabilă.", + "duplicateAutoComboFailedPrefix": "Duplicarea combinației automate a eșuat:", + "duplicateAutoComboUnknownError": "Eroare necunoscută", + "duplicateAutoComboTitle": "Creați o combinație statică din {name}" }, "costs": { "title": "Costuri", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index cf9ff5e8c26..8210e133a26 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -3722,7 +3722,12 @@ "errorDescription": "Мы не смогли загрузить данные комбо в данный момент. Проверьте ваше соединение и попробуйте снова.", "errorId": "Идентификатор ошибки: {id}", "errorRetry": "Попробуйте снова", - "comboLabel": "Комбо" + "comboLabel": "Комбо", + "duplicateAutoComboConfirm": "Создать статическое комбо из \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Это создаст снимок текущих подключенных провайдеров/моделей, соответствующих этому шаблону, в редактируемом комбо.", + "duplicateAutoComboFailedPrefix": "Не удалось дублировать автоматическое комбо:", + "duplicateAutoComboUnknownError": "Неизвестная ошибка", + "duplicateAutoComboTitle": "Создать статическое комбо из {name}" }, "costs": { "title": "Затраты", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index 10bf077a43c..ba4a09a0225 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -3722,7 +3722,12 @@ "errorDescription": "Momentálne sa nám nepodarilo načítať údaje kombinácie. Skontrolujte svoje pripojenie a skúste to znova.", "errorId": "Chyba ID: {id}", "errorRetry": "Skúste znova", - "comboLabel": "Kombinácia" + "comboLabel": "Kombinácia", + "duplicateAutoComboConfirm": "Vytvoriť staticú kombináciu z \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Tým sa zachytí aktuálne pripojení poskytovatelia/modely, ktoré zodpovedajú tejto šablóne, do upraviteľnej kombinácie.", + "duplicateAutoComboFailedPrefix": "Duplikácia automatickej kombinácie zlyhala:", + "duplicateAutoComboUnknownError": "Neznáma chyba", + "duplicateAutoComboTitle": "Vytvorte staticú kombináciu z {name}" }, "costs": { "title": "náklady", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 46f53972d7f..1433b5b49ce 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -3722,7 +3722,12 @@ "errorDescription": "Vi kunde inte ladda combo-data just nu. Kontrollera din anslutning och försök igen.", "errorId": "Fel-ID: {id}", "errorRetry": "Försök igen", - "comboLabel": "Kombination" + "comboLabel": "Kombination", + "duplicateAutoComboConfirm": "Skapa en statisk kombination från \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Detta kommer att ta en ögonblicksbild av de för närvarande anslutna leverantörerna/modellerna som matchar denna mall i en redigerbar kombination.", + "duplicateAutoComboFailedPrefix": "Kopiering av automatisk kombination misslyckades:", + "duplicateAutoComboUnknownError": "Okänt fel", + "duplicateAutoComboTitle": "Skapa en statisk kombination från {name}" }, "costs": { "title": "Kostnader", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 12d19b28c19..3a2b3c204ca 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -3722,7 +3722,12 @@ "errorDescription": "Hatuwezi kupakia data ya combo kwa sasa. Angalia muunganisho wako na ujaribu tena.", "errorId": "Kosa ID: {id}", "errorRetry": "Jaribu Tena", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Tengeneza combo thabiti kutoka \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Hii itachukua picha ya watoa huduma/watolei sambazwa sasa yanayolingana na kioo hiki katika combo inayoedit.", + "duplicateAutoComboFailedPrefix": "Imeshindwa kuiga combo ya otomatiki:", + "duplicateAutoComboUnknownError": "Hitilafai isiyojulikana", + "duplicateAutoComboTitle": "Tengeneza combo thabiti kutoka {name}" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 2e80cf7f22c..ab9fc3c1f32 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -3722,7 +3722,12 @@ "errorDescription": "நாங்கள் தற்போது கம்போ தரவுகளை ஏற்ற முடியவில்லை. உங்கள் இணைப்பை சரிபார்க்கவும் மற்றும் மீண்டும் முயற்சிக்கவும்.", "errorId": "பிழை அடையாளம்: {id}", "errorRetry": "மீண்டும் முயற்சி செய்", - "comboLabel": "கொம்போ" + "comboLabel": "கொம்போ", + "duplicateAutoComboConfirm": "\"{name}\" இலிருந்து நிலையான கம்போ உருவாக்கவா?", + "duplicateAutoComboSnapshotMsg": "இந்த டெம்ப்ளேட்டுடன் பொருந்தும் தற்போதைய இணைக்கப்பட்ட வழங்குநர்கள்/மாதிரிகளை திருத்தக்கூடிய கம்போவில் எடுக்கும்.", + "duplicateAutoComboFailedPrefix": "ஆட்டோகம்போ நகலெடுத்தல் தோல்வியடைந்தது:", + "duplicateAutoComboUnknownError": "அறியப்படாத பிழை", + "duplicateAutoComboTitle": "{name} இலிருந்து நிலையான கம்போ உருவாக்கவும்" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 73032e0d138..6d1a07e41b3 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -3722,7 +3722,12 @@ "errorDescription": "మేము ప్రస్తుతం కాంబో డేటాను లోడ్ చేయలేకపోయాము. మీ కనెక్షన్‌ను తనిఖీ చేసి మళ్లీ ప్రయత్నించండి.", "errorId": "లోపం ID: {id}", "errorRetry": "మరలా ప్రయత్నించండి", - "comboLabel": "కాంబో" + "comboLabel": "కాంబో", + "duplicateAutoComboConfirm": "\"{name}\" నుండి స్థిర కంబో సృష్టించాలా?", + "duplicateAutoComboSnapshotMsg": "ఈ టెంప్లేట్‌తో సరిపోయే ప్రస్తుత కనెక్ట్ చేసిన ప్రొవైడర్లు/మోడల్‌లను ఎడిటబుల్ కంబోలో స్నాప్‌షాట్ తీసుకుంటుంది.", + "duplicateAutoComboFailedPrefix": "ఆటోకంబో డూప్లికేట్ చేయడంలో విఫలమైంది:", + "duplicateAutoComboUnknownError": "తెలియని దోషం", + "duplicateAutoComboTitle": "{name} నుండి స్థిర కంబో సృష్టించండి" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 4b004659a95..b30a5bebc5e 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -3722,7 +3722,12 @@ "errorDescription": "ไม่สามารถโหลดข้อมูลคอมโบได้ในขณะนี้ กรุณาตรวจสอบการเชื่อมต่อของคุณและลองอีกครั้ง.", "errorId": "รหัสข้อผิดพลาด: {id}", "errorRetry": "ลองอีกครั้ง", - "comboLabel": "คอมโบ" + "comboLabel": "คอมโบ", + "duplicateAutoComboConfirm": "สร้างคอมโบคงที่จาก \"{name}\" หรือไม่?", + "duplicateAutoComboSnapshotMsg": "สิ่งนี้จะจับภาพผู้ให้บริการ/โมเดลที่เชื่อมต่ออยู่ในปัจจุบันซึ่งตรงกับเทมเพลตนี้ลงในคอมโบที่แก้ไขได้", + "duplicateAutoComboFailedPrefix": "ล้มเหลวในการทำสำเนาคอมโบอัตโนมัติ:", + "duplicateAutoComboUnknownError": "ข้อผิดพลาดที่ไม่ทราบสาเหตุ", + "duplicateAutoComboTitle": "สร้างคอมโบคงที่จาก {name}" }, "costs": { "title": "ค่าใช้จ่าย", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index abd8bfc297f..958b931040a 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -3722,7 +3722,12 @@ "errorDescription": "Şu anda kombinasyon verilerini yükleyemedik. Bağlantınızı kontrol edin ve tekrar deneyin.", "errorId": "Hata Kimliği: {id}", "errorRetry": "Tekrar Dene", - "comboLabel": "Kombinasyon" + "comboLabel": "Kombinasyon", + "duplicateAutoComboConfirm": "\"{name}\"-den statik bir kombinasyon oluşturulsun mu?", + "duplicateAutoComboSnapshotMsg": "Bu, bu şablona uygun olarak şu anda bağlı olan sağlayıcıları/modelleri düzenlenebilir bir kombinasyonda yakalayacaktır.", + "duplicateAutoComboFailedPrefix": "Otomatik kombinasyon kopyalanamadı:", + "duplicateAutoComboUnknownError": "Bilinmeyen hata", + "duplicateAutoComboTitle": "{name}-den statik bir kombinasyon oluştur" }, "costs": { "title": "Maliyetler", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index ce9ede8d176..586571a9ca3 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -3722,7 +3722,12 @@ "errorDescription": "Ми не змогли завантажити дані комбо прямо зараз. Перевірте своє з'єднання та спробуйте ще раз.", "errorId": "Ідентифікатор помилки: {id}", "errorRetry": "Спробуйте ще раз", - "comboLabel": "Комбо" + "comboLabel": "Комбо", + "duplicateAutoComboConfirm": "Створити статичне комбо з \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Це зробить знімок поточно підключених постачальників/моделей, які відповідають цьому шаблону, у редаговане комбо.", + "duplicateAutoComboFailedPrefix": "Не вдалося дублювати автоматичне комбо:", + "duplicateAutoComboUnknownError": "Невідома помилка", + "duplicateAutoComboTitle": "Створити статичне комбо з {name}" }, "costs": { "title": "Витрати", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index b5c237a7d53..aef5f7ae73e 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -3722,7 +3722,12 @@ "errorDescription": "ہم اس وقت کومبو ڈیٹا لوڈ نہیں کر سکے۔ اپنی کنکشن چیک کریں اور دوبارہ کوشش کریں۔", "errorId": "خرابی کی شناخت: {id}", "errorRetry": "پھر کوشش کریں", - "comboLabel": "کمبو" + "comboLabel": "کمبو", + "duplicateAutoComboConfirm": "\"{name}\" سے ایک جامد کمبو بنائیں؟", + "duplicateAutoComboSnapshotMsg": "یہ اس ٹیمپلیٹ سے ملنے والے موجودہ منسلک فراہم کنندگان/ماڈلز کو ایڈیٹ ایبل کمبو میں اسنیپ شاট لے گا۔", + "duplicateAutoComboFailedPrefix": "آٹو کمبو ڈپلیکیٹ کرنے میں ناکام:", + "duplicateAutoComboUnknownError": "نامعلوم خرابی", + "duplicateAutoComboTitle": "{name} سے ایک جامد کمبو بنائیں" }, "costs": { "title": "Costs", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 9cf64c0d2fa..0004b404b9a 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -3727,7 +3727,12 @@ "errorDescription": "Hiện không thể tải dữ liệu combo. Hãy kiểm tra kết nối rồi thử lại.", "errorId": "ID lỗi: {id}", "errorRetry": "Thử lại", - "comboLabel": "Combo" + "comboLabel": "Combo", + "duplicateAutoComboConfirm": "Tạo một tổ hợp tĩnh từ \"{name}\"?", + "duplicateAutoComboSnapshotMsg": "Điều này sẽ chụp lại các nhà cung cấp/mô hình đang kết nối hiện tại phù hợp với mẫu này vào một tổ hợp có thể chỉnh sửa.", + "duplicateAutoComboFailedPrefix": "Không thể sao chép tổ hợp tự động:", + "duplicateAutoComboUnknownError": "Lỗi không xác định", + "duplicateAutoComboTitle": "Tạo tổ hợp tĩnh từ {name}" }, "costs": { "title": "Chi phí", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index b4b6bad59bc..05eed0cd999 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -3722,7 +3722,12 @@ "errorDescription": "我们现在无法加载组合数据。请检查您的连接并重试。", "errorId": "错误 ID: {id}", "errorRetry": "再试一次", - "comboLabel": "组合" + "comboLabel": "组合", + "duplicateAutoComboConfirm": "从\"{name}\"创建静态组合?", + "duplicateAutoComboSnapshotMsg": "这将把与此模板匹配的当前连接提供者/模型快照到可编辑的组合中。", + "duplicateAutoComboFailedPrefix": "复制自动组合失败:", + "duplicateAutoComboUnknownError": "未知错误", + "duplicateAutoComboTitle": "从{name}创建静态组合" }, "costs": { "title": "成本", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index b2d96790866..442b9a66e74 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -3722,7 +3722,12 @@ "errorDescription": "目前無法加載組合數據。請檢查您的連接並重試。", "errorId": "錯誤 ID: {id}", "errorRetry": "再試一次", - "comboLabel": "組合" + "comboLabel": "組合", + "duplicateAutoComboConfirm": "從\"{name}\"建立靜態組合?", + "duplicateAutoComboSnapshotMsg": "這將把與此模板匹配的目前連線提供者/模型快照到可編輯的組合中。", + "duplicateAutoComboFailedPrefix": "複製自動組合失敗:", + "duplicateAutoComboUnknownError": "未知錯誤", + "duplicateAutoComboTitle": "從{name}建立靜態組合" }, "costs": { "title": "成本", diff --git a/src/lib/cliTools/checkToolConfigStatus.ts b/src/lib/cliTools/checkToolConfigStatus.ts index 75fcc5237a6..76d421c95a0 100644 --- a/src/lib/cliTools/checkToolConfigStatus.ts +++ b/src/lib/cliTools/checkToolConfigStatus.ts @@ -1,8 +1,12 @@ // DRY: shared between /api/cli-tools/status and /api/cli-tools/all-statuses (plan 14 F2) import fs from "fs/promises"; -import { getCliPrimaryConfigPath } from "@/shared/services/cliRuntime"; +import { getCliConfigHome, getCliPrimaryConfigPath } from "@/shared/services/cliRuntime"; import { hasOmniRouteQwenCodeConfig } from "@/shared/services/qwenCodeConfig"; +import { + parseGrokBuildConfig, + resolveGrokBuildConfigPath, +} from "@/shared/services/grokBuildConfig"; import { getRuntimePorts } from "@/lib/runtime/ports"; const { apiPort } = getRuntimePorts(); @@ -21,11 +25,24 @@ export async function checkToolConfigStatus( _configPathOverride?: string ): Promise<"configured" | "not_configured" | "not_installed" | "unknown" | "other"> { try { - const configPath = _configPathOverride ?? getCliPrimaryConfigPath(toolId); + const configPath = + _configPathOverride ?? + (toolId === "grok-build" + ? resolveGrokBuildConfigPath(process.env, getCliConfigHome()) + : getCliPrimaryConfigPath(toolId)); if (!configPath) return "unknown"; const content = await fs.readFile(configPath, "utf-8"); + if (toolId === "grok-build") { + const settings = parseGrokBuildConfig(content); + return settings.default === "omniroute" && + settings.model?.base_url && + settings.model.api_backend === "chat_completions" + ? "configured" + : "not_configured"; + } + // Codex uses TOML config — parse as raw text, not JSON if (toolId === "codex") { const lower = content.toLowerCase(); @@ -88,8 +105,8 @@ export async function checkToolConfigStatus( // (user may configure an external domain instead of localhost) if ( toolId === "cline" && - ((config.actModeApiProvider === "openai" || config.planModeApiProvider === "openai") && - ((config.openAiBaseUrl as string) || "").trim().length > 0) + (config.actModeApiProvider === "openai" || config.planModeApiProvider === "openai") && + ((config.openAiBaseUrl as string) || "").trim().length > 0 ) { return "configured"; } diff --git a/src/lib/db/callLogStats.ts b/src/lib/db/callLogStats.ts index 7df9908bd17..f3845f31f15 100644 --- a/src/lib/db/callLogStats.ts +++ b/src/lib/db/callLogStats.ts @@ -25,6 +25,15 @@ export interface ProviderMetricRow { lastErrorStatus: number | null; } +/** One provider's traffic over a bounded window. See `getProviderUsageSince`. */ +export interface ProviderUsageRow { + provider: string; + requests: number; + successes: number; + avgLatencyMs: number | null; + lastRequestAt: string | null; +} + export interface SearchProviderStatRow { provider: string; requests: number; @@ -107,6 +116,43 @@ export function getProviderMetrics(): ProviderMetricRow[] { .all() as ProviderMetricRow[]; } +// --------------------------------------------------------------------------- +// /api/free-provider-rankings — windowed usage aggregate +// --------------------------------------------------------------------------- + +/** + * Per-provider usage over a time window: how much traffic a provider actually + * served, and how much of it succeeded. + * + * Deliberately NOT `getProviderMetrics()` with a `since` parameter: that query + * carries two correlated subqueries (`lastStatus`, `lastErrorStatus`) which a + * ranking never displays, and they dominate its cost — `call_logs` is indexed + * on `timestamp` alone, so each correlated pass rescans the whole window per + * provider. Here a single bounded `GROUP BY` uses `idx_cl_timestamp` and stops + * there. The rules are shared with its neighbour, not the query: same success + * definition, same `#10714` guard against providers whose connections are gone. + */ +export function getProviderUsageSince(since: string): ProviderUsageRow[] { + const db = getDbInstance(); + return db + .prepare( + `SELECT + c.provider, + COUNT(*) as requests, + SUM(CASE WHEN c.status >= 200 AND c.status < 400 THEN 1 ELSE 0 END) as successes, + ROUND(AVG(c.duration)) as avgLatencyMs, + MAX(c.timestamp) as lastRequestAt + FROM call_logs c + WHERE c.provider IS NOT NULL AND c.provider != '-' + AND c.timestamp >= @since + AND EXISTS ( + SELECT 1 FROM provider_connections pc WHERE pc.provider = c.provider + ) + GROUP BY c.provider` + ) + .all({ since }) as ProviderUsageRow[]; +} + // --------------------------------------------------------------------------- // /api/search/stats — search provider aggregates + recent entries // --------------------------------------------------------------------------- diff --git a/src/lib/db/models/compat.ts b/src/lib/db/models/compat.ts index aa964fd2ef3..640f87b2649 100644 --- a/src/lib/db/models/compat.ts +++ b/src/lib/db/models/compat.ts @@ -1,6 +1,7 @@ /** db/models/compat.ts — model-compat overrides (normalizeToolCallId, per-protocol flags, upstream headers). */ import { getDbInstance } from "../core"; +import { resolveProviderAlias } from "@omniroute/open-sse/services/model.ts"; import { MODEL_COMPAT_PROTOCOL_KEYS, type ModelCompatProtocolKey, @@ -114,13 +115,17 @@ export type ModelCompatOverride = { compatByProtocol?: CompatByProtocolMap; upstreamHeaders?: Record; isHidden?: boolean; + apiFormat?: string; + targetFormat?: string; + supportsVision?: boolean; }; export function readCompatList(providerId: string): ModelCompatOverride[] { + const canonicalId = resolveProviderAlias(providerId) || providerId; const db = getDbInstance(); const row = db .prepare("SELECT value FROM key_value WHERE namespace = ? AND key = ?") - .get(MODEL_COMPAT_NAMESPACE, providerId); + .get(MODEL_COMPAT_NAMESPACE, canonicalId); const value = getKeyValue(row).value; if (!value) return []; try { @@ -140,16 +145,17 @@ export function readCompatList(providerId: string): ModelCompatOverride[] { } export function writeCompatList(providerId: string, list: ModelCompatOverride[]) { + const canonicalId = resolveProviderAlias(providerId) || providerId; const db = getDbInstance(); if (list.length === 0) { db.prepare("DELETE FROM key_value WHERE namespace = ? AND key = ?").run( MODEL_COMPAT_NAMESPACE, - providerId + canonicalId ); } else { db.prepare("INSERT OR REPLACE INTO key_value (namespace, key, value) VALUES (?, ?, ?)").run( MODEL_COMPAT_NAMESPACE, - providerId, + canonicalId, JSON.stringify(list) ); } @@ -168,6 +174,9 @@ export type ModelCompatPatch = { /** Replace top-level extra headers for override-only rows; omit to leave unchanged. */ upstreamHeaders?: Record | null; isHidden?: boolean | null; + apiFormat?: string | null; + targetFormat?: string | null; + supportsVision?: boolean | null; }; export function compatByProtocolHasEntries(map: CompatByProtocolMap | undefined): boolean { @@ -230,12 +239,39 @@ export function mergeModelCompatOverride( next.isHidden = Boolean(patch.isHidden); } } + if ("apiFormat" in patch) { + if (!patch.apiFormat) { + delete next.apiFormat; + } else { + next.apiFormat = patch.apiFormat; + } + } + if ("targetFormat" in patch) { + if (!patch.targetFormat) { + delete next.targetFormat; + } else { + next.targetFormat = patch.targetFormat; + } + } + if ("supportsVision" in patch) { + if (patch.supportsVision === null) { + delete next.supportsVision; + } else { + next.supportsVision = Boolean(patch.supportsVision); + } + } const hasHiddenFlag = Object.prototype.hasOwnProperty.call(next, "isHidden"); + const hasApiFormat = Object.prototype.hasOwnProperty.call(next, "apiFormat"); + const hasTargetFormat = Object.prototype.hasOwnProperty.call(next, "targetFormat"); + const hasVisionFlag = Object.prototype.hasOwnProperty.call(next, "supportsVision"); if ( next.normalizeToolCallId || hasPreserveFlag || hasVideoUrlFlag || hasHiddenFlag || + hasApiFormat || + hasTargetFormat || + hasVisionFlag || compatByProtocolHasEntries(next.compatByProtocol) || hasTopUpstream ) { diff --git a/src/lib/db/models/syncedAvailableModelPersistence.ts b/src/lib/db/models/syncedAvailableModelPersistence.ts index 7e8d13a1429..5f7257ff2c3 100644 --- a/src/lib/db/models/syncedAvailableModelPersistence.ts +++ b/src/lib/db/models/syncedAvailableModelPersistence.ts @@ -7,6 +7,33 @@ import { getKeyValue } from "./shared"; type ModelNormalizer = (models: unknown) => T[]; +// #11016: the Feature 5004 reconciler copies provider-declared windows +// (`inputTokenLimit`, captured at /models discovery) into `auto:discovery` +// overrides — the only source the REQUEST-TIME token-limit chain trusts for a +// freshly synced model that models.dev has not indexed yet. Before this, the +// reconcile ran only at startup + every 24h, so a model synced mid-cycle was +// enforced at the provider's static `defaultContextLength` (128K for +// OpenRouter) for up to a day even though the catalog advertised its real +// window (measured: `openrouter/stealth/ox-alpha` advertised 1,048,576, +// enforced 128,000). Run the reconcile opportunistically right after a synced +// catalog write changes, debounced + fire-and-forget so the sync request is +// never blocked (dynamic import: `contextWindowResolver` reads this module via +// `getAllSyncedAvailableModels`, so a static import would be circular). +let reconcileAfterSyncTimer: ReturnType | null = null; + +function scheduleReconcileAfterSyncWrite(): void { + if (reconcileAfterSyncTimer) return; // debounce bursty multi-connection writes + reconcileAfterSyncTimer = setTimeout(() => { + reconcileAfterSyncTimer = null; + void import("../../contextWindowResolver") + .then((m) => m.runContextWindowReconcile()) + .catch(() => { + // Swallow — the periodic reconcile still runs; sync must never fail on it. + }); + }, 0); + reconcileAfterSyncTimer.unref?.(); +} + export function finishSyncedAvailableModelsWrite(): void { backupDbFile("pre-write"); invalidateModelCatalogCache(); @@ -48,5 +75,6 @@ export function persistCanonicalSyncedAvailableModels( ).run(key, JSON.stringify(normalizedModels)); } finishSyncedAvailableModelsWrite(); + scheduleReconcileAfterSyncWrite(); return true; } diff --git a/src/lib/db/proxyLogs.ts b/src/lib/db/proxyLogs.ts index 033945d9d22..bb1e6ea7b3e 100644 --- a/src/lib/db/proxyLogs.ts +++ b/src/lib/db/proxyLogs.ts @@ -33,3 +33,37 @@ export function exportProxyLogsSince(since: string): Record[] { ); return stmt.all({ since }) as Record[]; } + +// 24h window for "last known egress IP" lookups. This helper answers a +// different question from proxyEgress.ts (#10677): that module reports which +// connections share an egress IP *right now*, derived from their proxy config +// and a live probe (5 min cache), while the lock needs the IP a connection +// actually *left through* on its recent traffic — history, which only +// proxy_logs holds. Hence a local window constant rather than a dependency. +// Exported so callers can build `since` without duplicating the window. +export const EGRESS_IP_LOOKUP_WINDOW_MS = 24 * 60 * 60 * 1000; + +/** + * Last non-null egress IP observed for a connection within the window, or + * null. Best-effort by design: egress_ip is only populated once the egress IP + * has been probed (cache TTL 5 min), so a cold cache yields null and the + * caller must fall back to today's behavior. Synchronous read (#10539 — no + * in-memory cache to go stale). The table has no index on connection_id + * (migration 134, YAGNI); the scan is bounded by the window via + * idx_pl_timestamp and this helper only runs at 429 frequency. + */ +export function getRecentEgressIpForConnection( + connectionId: string, + since: string +): { egressIp: string; at: string } | null { + const db = getDbInstance(); + const row = db + .prepare( + `SELECT egress_ip, timestamp FROM proxy_logs + WHERE connection_id = ? AND egress_ip IS NOT NULL AND timestamp >= ? + ORDER BY timestamp DESC LIMIT 1` + ) + .get(connectionId, since) as { egress_ip: string; timestamp: string } | undefined; + if (!row) return null; + return { egressIp: row.egress_ip, at: row.timestamp }; +} diff --git a/src/lib/db/settings.ts b/src/lib/db/settings.ts index edccc2da85c..ce2b8890661 100644 --- a/src/lib/db/settings.ts +++ b/src/lib/db/settings.ts @@ -162,6 +162,7 @@ export async function getSettings() { antigravitySignatureCacheMode: "enabled", requireLogin: true, oidcEnabled: false, + oidcDisablePasswordLogin: false, oidcIssuer: "", oidcClientId: "", oidcClientSecret: "", diff --git a/src/lib/evals/evalRunner.ts b/src/lib/evals/evalRunner.ts index d67ea9d74b0..d0b92800136 100644 --- a/src/lib/evals/evalRunner.ts +++ b/src/lib/evals/evalRunner.ts @@ -9,6 +9,7 @@ */ import { getCustomEvalSuite, listCustomEvalSuites } from "@/lib/db/evals"; +import safeRegex from "safe-regex"; import { goldenSet, codingSuite, @@ -161,6 +162,16 @@ export function evaluateCase(evalCase: any, actualOutput: string) { details.error = "Regex pattern too large for safe evaluation."; break; } + // G7 (silent-stop fix): a catastrophic regex (nested quantifiers like + // `(a+)+$`) can hang the event loop for minutes on adversarial output — + // the eval loop then "stops doing anything" with no error. safe-regex + // statically rejects such patterns before test() runs. + if (!safeRegex(regex)) { + passed = false; + details.error = + "Regex pattern rejected as potentially unsafe (catastrophic backtracking risk). Simplify the pattern."; + break; + } passed = regex.test(actualOutput); details.pattern = String(expectedValue); break; diff --git a/src/lib/freeProviderRankings.ts b/src/lib/freeProviderRankings.ts index 133a11ba676..37c99095c63 100644 --- a/src/lib/freeProviderRankings.ts +++ b/src/lib/freeProviderRankings.ts @@ -12,13 +12,22 @@ import { NOAUTH_PROVIDERS, OAUTH_PROVIDERS, APIKEY_PROVIDERS } from "@/shared/co import { REGISTRY } from "@omniroute/open-sse/config/providerRegistry"; import { listModelIntelligence } from "./db/modelIntelligence"; import { getProviderConnections } from "./db/providers"; +import { getProviderUsageSince, type ProviderUsageRow } from "./db/callLogStats"; import { getCustomModels } from "./db/models"; +// Type-only: reuse the health vocabulary instead of forking it. +import { RANGE_MS } from "./monitoring/providerHealthMatrix"; +import type { + ProviderHealthState, + ProviderHealthMatrixRange, +} from "./monitoring/providerHealthMatrix"; import type { ProviderAuthType } from "./freeProviderRankingsAuthType"; // Re-exported for backward-compat / same-module ergonomics (#6915) — the // actual implementations live in `freeProviderRankingsAuthType.ts` (DB-free, // safe to import from "use client" pages; see that file's header comment). export type { ProviderAuthType } from "./freeProviderRankingsAuthType"; +// Re-exported for consumers of `reliability`; the definition stays in monitoring. +export type { ProviderHealthState } from "./monitoring/providerHealthMatrix"; export { filterRankingsByAuthType, sortRankingsAuthTypeFirst, @@ -43,6 +52,8 @@ export interface FreeProviderRanking { topModel: ProviderModelScore | null; averageScore: number; modelCount: number; + /** Present only when connection state was loaded (filters active). See `ProviderReliability`. */ + reliability?: ProviderReliability; } /** @@ -222,6 +233,51 @@ export interface ConnectionState { rateLimitedUntil?: string | null; } +/** + * Second, additive dimension exposed on each ranking when connection state is + * loaded (configured/available filters active). Derived from data the ranking + * builder already holds — zero extra query. + * + * States use `ProviderHealthState` (`src/lib/monitoring/providerHealthMatrix.ts`) + * so both surfaces describe a provider the same way. The raw signals stay + * verbatim next to the state: `testStatus` is written on failure paths only and + * reset to `active` by an explicit connection test or a re-auth, so it can + * outlive the actual recovery. + */ +export interface ProviderReliability { + /** Same triplet `ProviderHealthMatrixAccount` exposes, one per connection. */ + connections: Array<{ + testStatus: string | null; + rateLimitedUntil: string | null; + state: ProviderHealthState; + }>; + /** Provider aggregate; absent entirely for providers with no loaded connection. */ + state: ProviderHealthState; + /** + * What the provider actually served over a window, from `call_logs`. Present + * only when the caller asks for it (`withUsage`). Complements `state`, which + * describes the connection right now and cannot see a provider that answers + * every call with an error. + */ + usage?: ProviderUsage; +} + +export interface ProviderUsage { + requests: number; + successes: number; + /** `null` below `MIN_USAGE_REQUESTS` — too small a sample to state a rate. */ + successRate: number | null; + avgLatencyMs: number | null; + lastRequestAt: string | null; + windowHours: number; +} + +/** + * Below this many requests in the window, no rate is reported: 1 failure out of + * 2 calls is not "50% broken", and a provider nobody called is not "0% healthy". + */ +const MIN_USAGE_REQUESTS = 5; + /** * Options controlling the additive "configured" / "available" filters. * Both default off (undefined/false) → output identical to current behavior. @@ -231,6 +287,30 @@ export interface FreeProviderRankingFilterOptions { configuredOnly?: boolean; /** Keep only providers that have ≥1 non-exhausted, non-rate-limited connection (implies configured). */ availableOnly?: boolean; + /** + * Also report what each provider actually served (`reliability.usage`). + * Off by default: it costs one aggregate query over `call_logs`, which a + * caller that only needs the ranking should not pay. + */ + withUsage?: boolean; + /** Window for `withUsage`. Defaults to `24h`, the health matrix's own default. */ + usageRange?: ProviderHealthMatrixRange; +} + +/** Group connection states by provider id (shared by filter and reliability attach). */ +function groupConnectionsByProvider( + connections: ConnectionState[] +): Map { + const byProvider = new Map(); + for (const conn of connections) { + const list = byProvider.get(conn.provider); + if (list) { + list.push(conn); + } else { + byProvider.set(conn.provider, [conn]); + } + } + return byProvider; } // Terminal connection statuses — mirrors `isTerminalConnectionStatus` @@ -250,16 +330,34 @@ const TERMINAL_CONNECTION_STATUSES = new Set(["credits_exhausted", "banned", "ex * quota lockout (model lockout, `open-sse/services/accountFallback.ts`) is a * deferred Phase 3 and is intentionally NOT consulted here. */ -export function isProviderUsable(connections: ConnectionState[], now: number = Date.now()): boolean { - return connections.some((conn) => { - const status = (conn.testStatus || "").trim().toLowerCase(); - if (TERMINAL_CONNECTION_STATUSES.has(status)) return false; - if (conn.rateLimitedUntil) { - const until = new Date(conn.rateLimitedUntil).getTime(); - if (Number.isFinite(until) && until > now) return false; - } - return true; - }); +export function isProviderUsable( + connections: ConnectionState[], + now: number = Date.now() +): boolean { + return connections.some((conn) => classifyConnection(conn, now) === "healthy"); +} + +/** + * One connection, classified as `classifyAccount` does (health matrix): terminal + * status ⇒ `down`, live cooldown ⇒ `degraded`, else `healthy`. Model lockouts are + * not loaded here, so — as in `isProviderUsable` — they are not consulted. + * The filter reuses this, so it cannot drift from the reported state. + */ +function classifyConnection(conn: ConnectionState, now: number): ProviderHealthState { + const status = (conn.testStatus || "").trim().toLowerCase(); + if (TERMINAL_CONNECTION_STATUSES.has(status)) return "down"; + if (conn.rateLimitedUntil) { + const until = new Date(conn.rateLimitedUntil).getTime(); + if (Number.isFinite(until) && until > now) return "degraded"; + } + return "healthy"; +} + +/** Mirrors `classifyProvider`, minus its circuit-breaker input (not loaded here). */ +function classifyProviderConnections(states: ProviderHealthState[]): ProviderHealthState { + if (states.length > 0 && states.every((state) => state === "down")) return "down"; + if (states.some((state) => state !== "healthy")) return "degraded"; + return "healthy"; } /** @@ -281,15 +379,7 @@ export function filterFreeProviderRankings( const { configuredOnly, availableOnly } = opts; if (!configuredOnly && !availableOnly) return rankings; - const byProvider = new Map(); - for (const conn of connections) { - const list = byProvider.get(conn.provider); - if (list) { - list.push(conn); - } else { - byProvider.set(conn.provider, [conn]); - } - } + const byProvider = groupConnectionsByProvider(connections); return rankings.filter((ranking) => { const conns = byProvider.get(ranking.id); @@ -299,6 +389,65 @@ export function filterFreeProviderRankings( }); } +/** + * Pure enrichment: attach `reliability` to every ranking with a loaded + * connection. Rankings without one are returned unchanged, never mutated. + */ +export function attachProviderReliability( + rankings: FreeProviderRanking[], + connections: ConnectionState[], + now: number = Date.now() +): FreeProviderRanking[] { + const byProvider = groupConnectionsByProvider(connections); + return rankings.map((ranking) => { + const conns = byProvider.get(ranking.id); + if (!conns || conns.length === 0) return ranking; + const states = conns.map((c) => classifyConnection(c, now)); + return { + ...ranking, + reliability: { + connections: conns.map((c, i) => ({ + testStatus: c.testStatus ?? null, + rateLimitedUntil: c.rateLimitedUntil ?? null, + state: states[i], + })), + state: classifyProviderConnections(states), + }, + }; + }); +} + +/** + * Pure enrichment: attach `usage` to the `reliability` of every ranking that has + * a row in the windowed aggregate. Rankings without `reliability` (no connection + * loaded) are returned unchanged, never mutated. + */ +export function attachProviderUsage( + rankings: FreeProviderRanking[], + usageRows: ProviderUsageRow[], + windowHours: number +): FreeProviderRanking[] { + const byProvider = new Map(usageRows.map((row) => [row.provider, row])); + return rankings.map((ranking) => { + const row = byProvider.get(ranking.id); + if (!row || !ranking.reliability) return ranking; + return { + ...ranking, + reliability: { + ...ranking.reliability, + usage: { + requests: row.requests, + successes: row.successes, + successRate: row.requests >= MIN_USAGE_REQUESTS ? row.successes / row.requests : null, + avgLatencyMs: row.avgLatencyMs ?? null, + lastRequestAt: row.lastRequestAt ?? null, + windowHours, + }, + }, + }; + }); +} + /** * Compute rankings for free providers based on ELO scores. * @@ -388,6 +537,24 @@ export async function computeFreeProviderRankings( isActive: true, })) as unknown as ConnectionState[]; filtered = filterFreeProviderRankings(rankings, connections, opts); + // Second dimension, same snapshot: annotation only, sort and scores untouched. + // `availableOnly` already drops providers with no healthy connection, so under + // it `state` is never `down`; `down` needs `configuredOnly` alone. + filtered = attachProviderReliability(filtered, connections); + + // Third dimension, opt-in: what the provider actually served. `state` above + // reads the connection as it stands now and cannot see a provider that + // answers every call with an error — only the call log can. + if (opts.withUsage) { + const range = opts.usageRange ?? "24h"; + const windowMs = RANGE_MS[range]; + const since = new Date(Date.now() - windowMs).toISOString(); + filtered = attachProviderUsage( + filtered, + getProviderUsageSince(since), + windowMs / (60 * 60 * 1000) + ); + } } return filtered.slice(0, limit); diff --git a/src/lib/guardrails/visionBridge.ts b/src/lib/guardrails/visionBridge.ts index 85cd7925127..db560353869 100644 --- a/src/lib/guardrails/visionBridge.ts +++ b/src/lib/guardrails/visionBridge.ts @@ -419,12 +419,25 @@ export class VisionBridgeGuardrail extends BaseGuardrail { // callVisionModel may fall back internally to another vision model, and // keying by attempt would fragment the cache and leak router state into the // key. Intentional and stable — do not "fix" this to key per attempt. + // + // The prompt component is the BASE prompt (`config.prompt`), deliberately + // NOT the task-aware composed prompt (`composedPrompt`): the composed + // prompt appends the LAST user text, which changes on every conversation + // turn. Zoo Code / Claude Code clients resend the FULL transcript each + // turn, so a text-only follow-up still carries the turn-1 image in the + // history — the bridge re-enters the describe path — but if the key + // changed with each new turn it would miss the cache and re-call the + // vision model (e.g. mimo-v2.5) for byte-identical images. Keying on the + // stable base prompt makes an unchanged history image reuse the cached + // description; a genuinely NEW image has a different contentRef and still + // misses. The task-aware prompt is what the vision model actually receives + // on the first describe, so no description quality is lost. const cache = runtime.cacheEnabled ? getSharedBridgeCacheFor(runtime) : null; // Process all images in parallel using Promise.allSettled for fail-partial behavior const results = await Promise.allSettled( limitedParts.map(async (imagePart, i) => { - const key = cache ? bridgeCacheKey(imagePart.imageUrl, composedPrompt, config.model) : null; + const key = cache ? bridgeCacheKey(imagePart.imageUrl, config.prompt, config.model) : null; const cached = key && cache ? cache.get(key) : undefined; const description = cached ?? (await callVision(imagePart.imageUrl, describeConfig)); if (cached === undefined && key && cache) cache.set(key, description); diff --git a/src/lib/guardrails/visionBridgeCredentials.ts b/src/lib/guardrails/visionBridgeCredentials.ts index 6b060222c8f..1e63e471e7d 100644 --- a/src/lib/guardrails/visionBridgeCredentials.ts +++ b/src/lib/guardrails/visionBridgeCredentials.ts @@ -5,6 +5,9 @@ * (visionBridge.ts already imports getBestVisionModel from visionBridgeRouter.ts). */ +import { resolveProviderId } from "@/shared/constants/providers"; +import { isNoAuthProviderKey } from "@/shared/utils/noAuthProviders"; + /** * True when a provider connection can actually authenticate upstream. * `noauth` with no real API key is NOT usable (opencode-zen free tier often @@ -36,9 +39,14 @@ function hasOAuthCredential(connection: ProviderConnectionLike): boolean { ); } -export function isProviderConnectionUsable(connection: ProviderConnectionLike): boolean { +/** True when the connection row carries a terminal status (disabled/banned/expired). */ +export function hasTerminalConnectionStatus(connection: ProviderConnectionLike): boolean { const status = String(connection.testStatus || "").toLowerCase(); - if (TERMINAL_CONNECTION_STATUSES.has(status)) { + return TERMINAL_CONNECTION_STATUSES.has(status); +} + +export function isProviderConnectionUsable(connection: ProviderConnectionLike): boolean { + if (hasTerminalConnectionStatus(connection)) { return false; } @@ -78,22 +86,34 @@ function loadProvidersModule(): Promise { /** * Resolve whether `provider/model` has at least one usable active connection. * Returns `null` when the credential store is unavailable (unit tests / early boot). + * + * The provider prefix is resolved alias→canonical id before querying + * `provider_connections` (the column stores the id, e.g. "opencode" for the + * "oc" alias — #10702: an alias-keyed query returned zero rows and excluded + * every candidate). No-auth providers (NOAUTH_PROVIDERS) need no stored API + * key: their effective credential is the synthetic "noauth" connection, so + * an empty active set is usable for them (unlike keyed providers). A stored + * row with a terminal status (disabled/banned/expired) still blocks the + * provider; any other row is treated as usable (the key requirement does not + * apply — a noauth row carries no API key by design). */ export async function hasUsableCredentialsForModel(model: string): Promise { - const rawPrefix = typeof model === "string" ? model.split("/")[0]?.trim() : ""; - if (!rawPrefix) return null; + const rawProvider = typeof model === "string" ? model.split("/")[0]?.trim() : ""; + if (!rawProvider) return null; + const provider = resolveProviderId(rawProvider); + const isNoAuth = isNoAuthProviderKey(rawProvider, provider); try { const { getProviderConnections } = await loadProvidersModule(); - // The model ids this module receives use the PUBLIC ALIAS (PROVIDER_MODELS keys, - // e.g. "cmd" for command-code), but provider_connections.provider is always - // persisted under the raw registry id — resolve the alias first, matching every - // other credential-check path (open-sse/services/model.ts, sse/services/auth.ts). - const { resolveProviderId } = await import("@/shared/constants/providers"); - const provider = resolveProviderId(rawPrefix); const connections = await getProviderConnections({ provider, isActive: true }); if (!Array.isArray(connections)) return null; - // Empty active set is a definitive "no" only when the table is readable. - if (connections.length === 0) return false; + // Empty active set: keyed providers are definitively unusable; no-auth + // providers still work through the synthetic "noauth" connection. + if (connections.length === 0) return isNoAuth; + // No-auth rows store no API key (authType "noauth" + empty apiKey would + // fail the generic key check) — only a terminal status blocks them. + if (isNoAuth) { + return !connections.some((c: any) => hasTerminalConnectionStatus(c)); + } return connections.some((c: any) => isProviderConnectionUsable(c)); } catch { return null; diff --git a/src/lib/guardrails/visionBridgeHelpers.ts b/src/lib/guardrails/visionBridgeHelpers.ts index bd903806132..02c99f0e94e 100644 --- a/src/lib/guardrails/visionBridgeHelpers.ts +++ b/src/lib/guardrails/visionBridgeHelpers.ts @@ -526,6 +526,11 @@ function parseSseVisionBody(rawBody: string): unknown { if (typeof delta?.reasoning_content === "string" && delta.reasoning_content.length > 0) { reasoningParts.push(delta.reasoning_content); } + // opencode-routed gateways (e.g. mimo-v2.5-free) stream chain-of-thought in + // `delta.reasoning` instead of `reasoning_content` (#6623). + if (typeof delta?.reasoning === "string" && delta.reasoning.length > 0) { + reasoningParts.push(delta.reasoning); + } // Some providers put a full message (not a delta) in the final chunk. const message = choice?.message as Record | undefined; @@ -535,6 +540,9 @@ function parseSseVisionBody(rawBody: string): unknown { if (typeof message?.reasoning_content === "string" && message.reasoning_content.length > 0) { reasoningParts.push(message.reasoning_content); } + if (typeof message?.reasoning === "string" && message.reasoning.length > 0) { + reasoningParts.push(message.reasoning); + } // Anthropic-style streaming: `content_block_delta` with `delta.text`. if (Array.isArray(unwrapped.content)) { @@ -602,13 +610,17 @@ async function readVisionResponseBody(response: Response): Promise { /** * Extract the description text from an OpenAI-compatible vision response. - * Falls back to `reasoning_content` when `content` is empty — reasoning models - * (e.g. xiaomi/mimo-v2.5) can exhaust `max_tokens` on chain-of-thought and - * return `content: null` with a complete analysis in `reasoning_content`. + * Falls back to `reasoning_content` then `reasoning` when `content` is empty — + * reasoning models (e.g. xiaomi/mimo-v2.5, opencode/mimo-v2.5-free) can exhaust + * `max_tokens` on chain-of-thought and return `content: null` with a complete + * analysis in a reasoning field. opencode-routed gateways name that field + * `reasoning` rather than `reasoning_content` (#6623 / #10809). */ function extractOpenAICompatibleContent(data: unknown): string { const record = data as { - choices?: Array<{ message?: { content?: unknown; reasoning_content?: unknown } }>; + choices?: Array<{ + message?: { content?: unknown; reasoning_content?: unknown; reasoning?: unknown }; + }>; error?: { message?: string }; } | null; @@ -625,7 +637,11 @@ function extractOpenAICompatibleContent(data: unknown): string { if (content) return content; const reasoning = - typeof message?.reasoning_content === "string" ? message.reasoning_content.trim() : ""; + typeof message?.reasoning_content === "string" + ? message.reasoning_content.trim() + : typeof message?.reasoning === "string" + ? message.reasoning.trim() + : ""; if (reasoning) return reasoning; throw new Error("Vision API returned empty or invalid response"); diff --git a/src/lib/kimi/tokenRefresh.ts b/src/lib/kimi/tokenRefresh.ts new file mode 100644 index 00000000000..213b3441ca3 --- /dev/null +++ b/src/lib/kimi/tokenRefresh.ts @@ -0,0 +1,107 @@ +import { getProviderConnectionById, updateProviderConnection } from "@/lib/db/providers"; +import { getKimiWebBaseUrl } from "@omniroute/open-sse/executors/kimi-web.ts"; +import { parseKimiJwt } from "@omniroute/open-sse/utils/kimiJwt.ts"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; + +export interface KimiRefreshResult { + success: boolean; + accessToken?: string; + refreshToken?: string; + expiresAtSec?: number; + error?: string; +} + +export async function exchangeKimiRefreshToken( + refreshToken: string, + baseUrl?: string +): Promise { + const cleanRefresh = String(refreshToken ?? "").trim(); + if (!cleanRefresh) { + return { success: false, error: "No refresh_token provided" }; + } + + const effectiveBaseUrl = (baseUrl || getKimiWebBaseUrl()).replace(/\/+$/, ""); + const refreshUrl = `${effectiveBaseUrl}/api/auth/token/refresh`; + + try { + const resp = await fetch(refreshUrl, { + method: "GET", + headers: { + Authorization: `Bearer ${cleanRefresh}`, + Accept: "application/json, text/plain, */*", + Origin: effectiveBaseUrl, + Referer: `${effectiveBaseUrl}/`, + "User-Agent": + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", + }, + }); + + if (!resp.ok) { + const errText = await resp.text().catch(() => ""); + return { + success: false, + error: `Kimi refresh returned HTTP ${resp.status}: ${sanitizeErrorMessage(errText)}`, + }; + } + + const data = await resp.json(); + const newAccess = data?.access_token; + const newRefresh = data?.refresh_token || cleanRefresh; + + if (!newAccess || typeof newAccess !== "string") { + return { success: false, error: "Invalid response from Kimi: missing access_token" }; + } + + const parsedJwt = parseKimiJwt(newAccess); + const expiresAtSec = parsedJwt?.exp || Math.floor(Date.now() / 1000) + 900; + + return { + success: true, + accessToken: newAccess, + refreshToken: newRefresh, + expiresAtSec, + }; + } catch (err) { + return { + success: false, + error: `Network error refreshing Kimi token: ${err instanceof Error ? err.message : "unknown"}`, + }; + } +} + +export async function refreshKimiProviderConnection( + connectionId: string +): Promise { + const conn = await getProviderConnectionById(connectionId); + if (!conn) { + return { success: false, error: `Connection ${connectionId} not found` }; + } + + const providerData = conn.providerSpecificData as Record | undefined; + const refreshToken: string = + typeof conn.refreshToken === "string" && conn.refreshToken + ? conn.refreshToken + : typeof providerData?.refreshToken === "string" + ? providerData.refreshToken + : ""; + if (!refreshToken) { + return { success: false, error: "Connection does not contain a refresh_token" }; + } + + const result = await exchangeKimiRefreshToken(refreshToken); + if (!result.success || !result.accessToken) { + return result; + } + + await updateProviderConnection(connectionId, { + apiKey: result.accessToken, + accessToken: result.accessToken, + refreshToken: result.refreshToken, + expiresAt: result.expiresAtSec ? new Date(result.expiresAtSec * 1000).toISOString() : undefined, + testStatus: "active", + lastError: null, + errorCode: null, + }); + + return result; +} diff --git a/src/lib/modelCapabilities.ts b/src/lib/modelCapabilities.ts index c0b20dfb59a..727122d2048 100644 --- a/src/lib/modelCapabilities.ts +++ b/src/lib/modelCapabilities.ts @@ -439,10 +439,19 @@ export function modelIdLikelyVision(modelId: string | null | undefined): boolean * models are text-only (mimo.mi.com .../image-understanding; hermes-agent#18884). * Anchored to the full id (`$`) and tolerant of a `provider/` prefix so `mimo-v2.5-pro` * never matches the multimodal `mimo-v2.5`, and `mimo-v2-pro` never matches `mimo-v2-omni`. + * + * Command Code `cmd/gpt-5.3-codex*` (#10703): the Command Code registry marks + * `gpt-5.3-codex` as `supportsVision: true`, but the gateway actually exposes + * it as a text-only code model — selecting it as a Vision Bridge candidate + * failed every image describe call (#10703, CONTRIBUTOR-reported). Scoped to + * the command-code alias/id so only the Command-Code-gateway Codex variants + * are overridden; genuine multimodal `gpt-5.x` chat models (e.g. `gpt-5.5`, + * `gpt-5.4-mini`, real OpenAI `openai/gpt-5.3-codex`) keep their vision verdict. */ const KNOWN_TEXT_ONLY_DESPITE_SYNC: readonly RegExp[] = [ /(?:^|\/)mimo-v2\.5-pro$/i, /(?:^|\/)mimo-v2-pro$/i, + /^(?:cmd|command-code)\/gpt-5\.3-codex(?:-|$)/i, ]; function isKnownTextOnlyDespiteSync(modelId: string | null | undefined): boolean { @@ -576,7 +585,10 @@ function getContextOverride( * `snapshot` is the #9147 build-local bulk load; when supplied the on-demand * SQLite read is skipped and the preloaded nested map is used instead. */ -export function getResolvedModelContextOverride(input: CapabilityInput, snapshot?: ModelCapabilityResolutionSnapshot | null): number | null { +export function getResolvedModelContextOverride( + input: CapabilityInput, + snapshot?: ModelCapabilityResolutionSnapshot | null +): number | null { return getContextOverride(resolveCapabilityInput(input), snapshot); } diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 5e477084ea3..3828047b90b 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -457,6 +457,31 @@ export function enrichCatalogModelEntry( { provider, model }, capabilitySnapshot ); + const existingCapabilities = + entry.capabilities && typeof entry.capabilities === "object" + ? (entry.capabilities as JsonRecord) + : {}; + const declaredEffortTiers = Array.isArray(existingCapabilities.effort_tiers) + ? existingCapabilities.effort_tiers.filter( + (effort): effort is string => typeof effort === "string" && effort.length > 0 + ) + : []; + const sourceDeclaresThinking = + typeof existingCapabilities.thinking === "boolean" || + typeof existingCapabilities.supportsThinking === "boolean"; + const effortTiers = + metadata.capabilities.supportedThinkingEfforts && + metadata.capabilities.supportedThinkingEfforts.length > 0 + ? [...metadata.capabilities.supportedThinkingEfforts] + : declaredEffortTiers.length > 0 + ? declaredEffortTiers + : sourceDeclaresThinking + ? undefined + : extendCodexGpt56EffortValues( + metadata.provider, + metadata.model, + CANONICAL_EFFORT_VALUES + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -482,18 +507,8 @@ export function enrichCatalogModelEntry( ? { thinking: metadata.capabilities.supportsThinking, supportsThinking: metadata.capabilities.supportsThinking, - ...(metadata.capabilities.supportsThinking - ? { - effort_tiers: - metadata.capabilities.supportedThinkingEfforts && - metadata.capabilities.supportedThinkingEfforts.length > 0 - ? [...metadata.capabilities.supportedThinkingEfforts] - : extendCodexGpt56EffortValues( - metadata.provider, - metadata.model, - CANONICAL_EFFORT_VALUES - ), - } + ...(metadata.capabilities.supportsThinking && effortTiers + ? { effort_tiers: effortTiers } : {}), } : {}), @@ -509,9 +524,7 @@ export function enrichCatalogModelEntry( }; nextEntry.capabilities = { - ...(entry.capabilities && typeof entry.capabilities === "object" - ? (entry.capabilities as JsonRecord) - : {}), + ...existingCapabilities, ...capabilityFields, }; @@ -548,15 +561,30 @@ export function enrichCatalogModelEntry( } const persistedOutputLimit = - getModelCapabilityOverride(provider, model, "max_output_tokens", capabilitySnapshot?.maxTokenOverrides) ?? - getModelCapabilityOverride(provider, model, "max_token", capabilitySnapshot?.maxTokenOverrides) ?? + getModelCapabilityOverride( + provider, + model, + "max_output_tokens", + capabilitySnapshot?.maxTokenOverrides + ) ?? + getModelCapabilityOverride( + provider, + model, + "max_token", + capabilitySnapshot?.maxTokenOverrides + ) ?? getModelCapabilityOverride( publicProvider, model, "max_output_tokens", capabilitySnapshot?.maxTokenOverrides ) ?? - getModelCapabilityOverride(publicProvider, model, "max_token", capabilitySnapshot?.maxTokenOverrides); + getModelCapabilityOverride( + publicProvider, + model, + "max_token", + capabilitySnapshot?.maxTokenOverrides + ); if (persistedOutputLimit !== null) { nextEntry.max_output_tokens = persistedOutputLimit; } else if ( diff --git a/src/lib/monitoring/providerHealthMatrix.ts b/src/lib/monitoring/providerHealthMatrix.ts index b1ff287baee..f740a7570cf 100644 --- a/src/lib/monitoring/providerHealthMatrix.ts +++ b/src/lib/monitoring/providerHealthMatrix.ts @@ -121,7 +121,8 @@ interface CallLogTargetStats { lastErrorStatus: number | null; } -const RANGE_MS: Record = { +/** Exported so other surfaces reporting over a window use the same scale. */ +export const RANGE_MS: Record = { "1h": 60 * 60 * 1000, "24h": 24 * 60 * 60 * 1000, "7d": 7 * 24 * 60 * 60 * 1000, diff --git a/src/lib/providerModels/modelDiscovery.ts b/src/lib/providerModels/modelDiscovery.ts index 5b41227f247..85e605daff4 100644 --- a/src/lib/providerModels/modelDiscovery.ts +++ b/src/lib/providerModels/modelDiscovery.ts @@ -68,6 +68,19 @@ export function detectVisionInput(record: JsonRecord): boolean { // import format already emits). Hard Rule #7 — validate the untrusted upstream // payload with Zod before it is trusted/stored; a malformed shape degrades to // `undefined` instead of throwing, so one bad record never fails the whole sync. +// The same nesting also carries `default_effort` (e.g. OpenRouter +// `reasoning:{mandatory, default_enabled, default_effort, supported_efforts}`) — +// captured by `detectDefaultThinkingEffort` below and threaded through the +// EXISTING `defaultThinkingEffort` plumbing (`SyncedAvailableModel`, +// RuntimeModelMeta, #6879 `applyDefaultReasoningEffort`), so a model that only +// produces usable output with an explicit effort (measured: OpenRouter stealth +// reasoning models returning `upstream_empty_response` without one) gets the +// vendor-declared default injected instead of failing. +const reasoningDefaultEffortSchema = z + .object({ default_effort: z.string().optional() }) + .partial() + .nullable() + .optional(); const reasoningSupportedEffortsSchema = z .object({ supported_efforts: z.array(z.string()).optional() }) .partial() @@ -98,6 +111,11 @@ const EFFORT_SYNONYMS: Record = { max: "xhigh" }; // Live request testing confirms Crof accepts `max` as a distinct top tier. const CROF_REASONING_EFFORTS = ["none", "low", "medium", "high", "max"] as const; +// Command Code's provider API accepts the documented low/medium/high/xhigh/max +// reasoning_effort values for its reasoning-capable model catalog, but its +// /models response does not declare them. Keep this fallback provider-scoped. +const COMMAND_CODE_REASONING_EFFORTS = ["low", "medium", "high", "xhigh", "max"] as const; + function normalizeSupportedEffort(effort: string): string { if ((CANONICAL_EFFORT_VALUES as readonly string[]).includes(effort)) return effort; return EFFORT_SYNONYMS[effort.toLowerCase()] || effort; @@ -130,6 +148,28 @@ function parseEffortList(rawList: unknown): string[] | undefined { return efforts.length > 0 ? efforts : undefined; } +/** + * Read the nested `record.reasoning.default_effort` shape (OpenRouter declares + * `reasoning:{mandatory, default_enabled, default_effort, supported_efforts}`) + * and normalize it onto the canonical vocabulary (`max` → `xhigh`, same mapping + * `detectSupportedThinkingEfforts` applies to the tier list). Returns `undefined` + * (never throws) when the field is absent or malformed. + * + * A flat top-level `defaultThinkingEffort` (OmniRoute's own import format, and + * kimi-style upstreams) stays authoritative — the nested shape is a fallback. + */ +export function detectDefaultThinkingEffort(record: JsonRecord): string | undefined { + if (typeof record.defaultThinkingEffort === "string" && record.defaultThinkingEffort.length > 0) { + return normalizeSupportedEffort(record.defaultThinkingEffort); + } + const parsed = reasoningDefaultEffortSchema.safeParse(record.reasoning); + if (parsed.success && parsed.data) { + const raw = parsed.data.default_effort; + if (typeof raw === "string" && raw.length > 0) return normalizeSupportedEffort(raw); + } + return undefined; +} + /** * #7694: read the nested `record.reasoning.supported_efforts` shape and normalize each * tier onto the canonical vocabulary. Returns `undefined` (never throws) when the field @@ -232,6 +272,7 @@ export function normalizeDiscoveredModels( if (!id) continue; const isCrofReasoningModel = providerId === "crof" && record.reasoning_effort === true; + const isCommandCodeModel = providerId === "command-code"; const supportedThinkingEfforts = (() => { // The flat import field and every recognized upstream tier array remain // authoritative over the provider fallback, including an explicit empty list. @@ -242,8 +283,12 @@ export function normalizeDiscoveredModels( } const detected = detectSupportedThinkingEfforts(record); if (detected || hasDeclaredEffortList(record)) return detected; - return isCrofReasoningModel ? [...CROF_REASONING_EFFORTS] : undefined; + if (isCrofReasoningModel) return [...CROF_REASONING_EFFORTS]; + return isCommandCodeModel ? [...COMMAND_CODE_REASONING_EFFORTS] : undefined; })(); + // Vendor-declared default effort (OpenRouter `reasoning.default_effort`, or the + // flat import field). Normalized onto the canonical vocabulary (`max` → `xhigh`). + const defaultThinkingEffort = detectDefaultThinkingEffort(record); const name = toNonEmptyString(record.name) || @@ -300,15 +345,13 @@ export function normalizeDiscoveredModels( : {}), ...(supportedEndpoints && supportedEndpoints.length > 0 ? { supportedEndpoints } : {}), ...(supportedThinkingEfforts !== undefined ? { supportedThinkingEfforts } : {}), - ...(toNonEmptyString(record.defaultThinkingEffort) - ? { defaultThinkingEffort: toNonEmptyString(record.defaultThinkingEffort)! } - : {}), + ...(defaultThinkingEffort !== undefined ? { defaultThinkingEffort } : {}), ...(typeof inputTokenLimit === "number" ? { inputTokenLimit } : {}), ...(typeof outputTokenLimit === "number" ? { outputTokenLimit } : {}), ...(typeof record.description === "string" ? { description: record.description } : {}), ...(typeof record.supportsThinking === "boolean" ? { supportsThinking: record.supportsThinking } - : isCrofReasoningModel + : isCrofReasoningModel || isCommandCodeModel ? { supportsThinking: true } : {}), ...(record.alwaysThinking === true ? { alwaysThinking: true } : {}), diff --git a/src/lib/providers/requestDefaults.ts b/src/lib/providers/requestDefaults.ts index 05786ab7aa4..b54bf3669d2 100644 --- a/src/lib/providers/requestDefaults.ts +++ b/src/lib/providers/requestDefaults.ts @@ -218,6 +218,14 @@ export function normalizeProviderSpecificData( delete normalized.autoFetchModels; } + // Per-connection operator timeout — only persist a real integer. + if ( + "timeoutMs" in normalized && + (typeof normalized.timeoutMs !== "number" || !Number.isInteger(normalized.timeoutMs)) + ) { + delete normalized.timeoutMs; + } + if ("preset" in normalized) { const preset = provider === "openrouter" ? normalizeOpenRouterPreset(normalized.preset) : null; if (preset) { diff --git a/src/lib/providers/webCookieAuth.ts b/src/lib/providers/webCookieAuth.ts index 0796e11fac8..2035ed444f0 100644 --- a/src/lib/providers/webCookieAuth.ts +++ b/src/lib/providers/webCookieAuth.ts @@ -170,6 +170,15 @@ export function extractKimiAccessToken(rawValue: string): string { const raw = String(rawValue ?? "").trim(); if (!raw) return ""; + // 1. JSON dump extraction + if (raw.startsWith("{") && raw.endsWith("}")) { + try { + const parsed = JSON.parse(raw); + const access = parsed?.access_token || parsed?.token || ""; + if (access && typeof access === "string") return access.trim(); + } catch {} + } + const bearer = raw.match(/^(?:authorization:\s*)?bearer\s+([^;\s]+)/i); if (bearer) return bearer[1]; @@ -183,6 +192,42 @@ export function extractKimiAccessToken(rawValue: string): string { return !trimmed.includes("=") && !trimmed.includes(";") ? trimmed : ""; } +/** Extract Kimi Web refresh_token from key-value, raw string, or localStorage JSON dump. */ +export function extractKimiRefreshToken(rawValue: string): string { + const raw = String(rawValue ?? "").trim(); + if (!raw) return ""; + + // 1. JSON dump extraction + if (raw.startsWith("{") && raw.endsWith("}")) { + try { + const parsed = JSON.parse(raw); + if (parsed?.refresh_token && typeof parsed.refresh_token === "string") { + return parsed.refresh_token.trim(); + } + } catch {} + } + + // 2. Key-value extraction + const match = raw.match(/(?:^|[\s;])refresh_token=([^;\s]+)/); + if (match) return match[1]; + + return ""; +} + +/** Extract both access_token and refresh_token from user input. */ +export function extractKimiCredentials(rawValue: string): { + accessToken: string; + refreshToken: string; +} { + const raw = String(rawValue ?? "").trim(); + if (!raw) return { accessToken: "", refreshToken: "" }; + + return { + accessToken: extractKimiAccessToken(raw), + refreshToken: extractKimiRefreshToken(raw), + }; +} + /** @deprecated Use extractKimiAccessToken; retained for existing imports. */ export function extractKimiJwt(rawValue: string): string { return extractKimiAccessToken(rawValue); diff --git a/src/lib/proxyRelay/cloudflareWorkerScript.ts b/src/lib/proxyRelay/cloudflareWorkerScript.ts index 73de16b9f25..c25c6021229 100644 --- a/src/lib/proxyRelay/cloudflareWorkerScript.ts +++ b/src/lib/proxyRelay/cloudflareWorkerScript.ts @@ -10,6 +10,12 @@ * - Inlines an SSRF guard rejecting RFC1918 / loopback / link-local / IPv6 ULA * targets — the Edge runtime cannot import Node helpers, the guard lives * here as a string. + * - Resolves x-relay-path through the SAME `resolveRelayTarget()` the Deno and + * Vercel workers use (PR #4643 and its follow-up), instead of concatenating + * it onto the target. Concatenation lets the path re-point the request past + * the validated host. Bound to a LITERAL const name so the hardcoded call + * site still resolves when the SWC-minified standalone build mangles the + * source function's own name in `.toString()` output (#6149). * - Strips Host + relay control headers before forwarding upstream. * * The string template is fed to Cloudflare's PUT /accounts/{id}/workers/scripts/{name} @@ -23,6 +29,8 @@ * - SSRF guard is inlined so a leaked relay URL cannot scan internal IPs. */ import { randomUUID } from "crypto"; +import { resolveRelayTarget } from "@/app/api/settings/proxy/deno-deploy/route"; +import { isPrivateRelayHostname } from "@/lib/proxyRelay/privateHostname"; /** * Build the multipart/form-data request body for Cloudflare's Worker @@ -75,33 +83,9 @@ export function buildCloudflareWorkerScript(relayAuth: string): string { // user-controlled input ever reaches this template, so direct interpolation // into the worker source string is safe. return `// OmniRoute Cloudflare Worker proxy relay — generated at deploy time. -function isPrivateHostname(h) { - if (!h) return true; - const host = h.trim().toLowerCase().replace(/^\\[|\\]$/g, ""); - if ( - host === "localhost" || - host === "0.0.0.0" || host === "127.0.0.1" || host === "::1" || - host.endsWith(".localhost") || - host.endsWith(".local") || - host.endsWith(".internal") || - host.startsWith("::ffff:") - ) return true; - const v4 = host.match(/^(\\d{1,3})\\.(\\d{1,3})\\.(\\d{1,3})\\.(\\d{1,3})$/); - if (v4) { - const a = +v4[1], b = +v4[2]; - if (a === 0 || a === 10 || a === 127) return true; - if (a === 169 && b === 254) return true; // link-local IPv4 - if (a === 192 && b === 168) return true; - if (a === 172 && b >= 16 && b <= 31) return true; - if (a === 100 && b >= 64 && b <= 127) return true; - return false; - } - if (host.includes(":")) { - // IPv6 loopback/ULA/link-local (fe80::/10) - return host === "::1" || host.startsWith("fc") || host.startsWith("fd") || host.startsWith("fe80:"); - } - return false; -} +const resolveRelayTarget = ${resolveRelayTarget.toString()}; + +const isPrivateHostname = ${isPrivateRelayHostname.toString()}; async function handleRelay(request) { const auth = request.headers.get("x-relay-auth"); @@ -134,9 +118,12 @@ async function handleRelay(request) { init.body = request.body; init.duplex = "half"; } + const resolved = resolveRelayTarget(target, relayPath); + if (!resolved.ok) { + return new Response(resolved.reason, { status: resolved.status }); + } try { - const targetBase = target.endsWith("/") ? target.slice(0, -1) : target; - const upstream = await fetch(targetBase + relayPath, init); + const upstream = await fetch(resolved.url, init); return new Response(upstream.body, { status: upstream.status, headers: upstream.headers, diff --git a/src/lib/proxyRelay/privateHostname.ts b/src/lib/proxyRelay/privateHostname.ts new file mode 100644 index 00000000000..7cc800baf09 --- /dev/null +++ b/src/lib/proxyRelay/privateHostname.ts @@ -0,0 +1,67 @@ +/** + * Shared private/loopback host guard for the three proxy-relay workers + * (Cloudflare, Deno Deploy, Vercel Edge). + * + * The three generators each carried a byte-identical copy of this policy inlined + * as a string, so a gap had to be found and fixed three times. It is now written + * once and embedded verbatim via `Function#toString`, the same mechanism + * `resolveRelayTarget` already uses — the edge runtimes cannot import Node + * helpers, so the source has to travel as text. + * + * Pure (only `String`/`RegExp`, no Node or Deno globals) so the SAME source runs + * in every worker and is unit-testable directly in Node. + * + * Callers pass `new URL(target).hostname`, which is already WHATWG-normalized: + * `2130706433` arrives as `127.0.0.1`, and `::ffff:127.0.0.1` arrives as + * `[::ffff:7f00:1]`. The brackets are stripped here. + */ +export function isPrivateRelayHostname(h: string): boolean { + if (!h) return true; + let host = String(h) + .trim() + .toLowerCase() + .replace(/^\[|\]$/g, ""); + // Drop the FQDN root dot. `localhost.` resolves exactly like `localhost`, and + // a trailing dot otherwise slips past every exact and suffix test below — + // including `.internal`, so `svc.internal.` would have been allowed. + if (host.length > 1 && host.endsWith(".")) host = host.slice(0, -1); + if (!host) return true; + + if ( + host === "localhost" || + host === "0.0.0.0" || + host === "127.0.0.1" || + host.endsWith(".localhost") || + host.endsWith(".local") || + host.endsWith(".internal") + ) { + return true; + } + + // Everything in ::/96 — the unspecified address, IPv6 loopback, IPv4-mapped + // (`::ffff:7f00:1`) and the deprecated IPv4-compatible form (`::7f00:1`). + // None of them is a legitimate public relay target, and `http://[::]/` reaches + // a service bound to the IPv6 loopback. + if (host.startsWith("::")) return true; + + const v4 = host.match(/^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/); + if (v4) { + const a = Number(v4[1]); + const b = Number(v4[2]); + if (a === 0 || a === 10 || a === 127) return true; + if (a === 169 && b === 254) return true; // link-local IPv4 + if (a === 192 && b === 168) return true; + if (a === 172 && b >= 16 && b <= 31) return true; + if (a === 100 && b >= 64 && b <= 127) return true; // CGNAT + return false; + } + + if (host.includes(":")) { + if (host.startsWith("fc") || host.startsWith("fd")) return true; // ULA fc00::/7 + // Link-local is fe80::/10 — fe80 through febf, not only the `fe80:` spelling. + if (/^fe[89ab]/.test(host)) return true; + return false; + } + + return false; +} diff --git a/src/lib/quota/providerCapabilities.ts b/src/lib/quota/providerCapabilities.ts deleted file mode 100644 index cc5ec32a767..00000000000 --- a/src/lib/quota/providerCapabilities.ts +++ /dev/null @@ -1,37 +0,0 @@ -export interface ProviderCapabilities { - providerId: string; - quotaApi: boolean; - usageApi: boolean; - rateLimitHeaders: boolean; - streaming: boolean; - toolUse: boolean; - coding: boolean; - vision: boolean; - longContext: boolean; -} - -const registry = new Map(); - -export function registerProviderCapabilities(capabilities: ProviderCapabilities): void { - registry.set(capabilities.providerId, { ...capabilities }); -} - -export function getProviderCapabilities(providerId: string): ProviderCapabilities { - return ( - registry.get(providerId) ?? { - providerId, - quotaApi: false, - usageApi: false, - rateLimitHeaders: false, - streaming: false, - toolUse: false, - coding: false, - vision: false, - longContext: false, - } - ); -} - -export function listProviderCapabilities(): ProviderCapabilities[] { - return [...registry.values()].map((capabilities) => ({ ...capabilities })); -} diff --git a/src/lib/quota/providerQuotaTelemetry.ts b/src/lib/quota/providerQuotaTelemetry.ts index cd9a29f2ad1..05955690c6f 100644 --- a/src/lib/quota/providerQuotaTelemetry.ts +++ b/src/lib/quota/providerQuotaTelemetry.ts @@ -48,12 +48,6 @@ export interface ProviderConnectionForQuota { [key: string]: unknown; } -export interface ProviderQuotaMonitor { - providerId: string; - supportedDimensions(): Promise; - fetchQuotaState(connection: ProviderConnectionForQuota): Promise; -} - export type QuotaSourceKind = "provider_api" | "response_headers" | "configured" | "estimated" | "unknown"; diff --git a/src/lib/skills/registry.ts b/src/lib/skills/registry.ts index fd39e655548..4621ff9a6d8 100644 --- a/src/lib/skills/registry.ts +++ b/src/lib/skills/registry.ts @@ -1,4 +1,4 @@ -import { Skill, SkillSchema } from "./types"; +import type { Skill, SkillSchema } from "./types"; import { SkillCreateInputSchema } from "./schemas"; import { getDbInstance } from "../db/core"; import { randomUUID } from "crypto"; @@ -6,6 +6,10 @@ import { logger } from "../../../open-sse/utils/logger.ts"; const log = logger("SKILLS"); +export const GLOBAL_SKILL_OWNER_ID = "system"; +const GLOBAL_SKILL_OWNER_IDS = [GLOBAL_SKILL_OWNER_ID, "skillsmp", "skillssh"] as const; +const GLOBAL_SKILL_OWNER_ID_SET = new Set(GLOBAL_SKILL_OWNER_IDS); + class SkillRegistry { private static instance: SkillRegistry; private registeredSkills: Map = new Map(); @@ -39,6 +43,36 @@ class SkillRegistry { return `${skill.apiKeyId}:${skill.name}@${skill.version}`; } + private skillIdentity(skill: Pick): string { + return `${skill.name}@${skill.version}`; + } + + private isGlobalOwner(apiKeyId: string): boolean { + return GLOBAL_SKILL_OWNER_ID_SET.has(apiKeyId); + } + + private scopedSkills(apiKeyId?: string): Skill[] { + const skills = Array.from(this.registeredSkills.values()); + if (!apiKeyId) return skills; + + const owned = skills.filter((skill) => skill.apiKeyId === apiKeyId); + const visibleIdentities = new Set(owned.map((skill) => this.skillIdentity(skill))); + const global = [ + ...skills.filter((skill) => skill.apiKeyId === GLOBAL_SKILL_OWNER_ID), + ...skills.filter( + (skill) => skill.apiKeyId !== GLOBAL_SKILL_OWNER_ID && this.isGlobalOwner(skill.apiKeyId) + ), + ]; + for (const skill of global) { + const identity = this.skillIdentity(skill); + if (!visibleIdentities.has(identity)) { + owned.push(skill); + visibleIdentities.add(identity); + } + } + return owned; + } + private cacheSkill(skill: Skill): void { this.registeredSkills.set(this.cacheKey(skill), skill); this.updateVersionCache(skill); @@ -180,16 +214,11 @@ class SkillRegistry { list(apiKeyId?: string): Skill[] { log.debug("skills.registry.list", { apiKeyId, cached: !this.isCacheStale() }); - if (apiKeyId) { - return Array.from(this.registeredSkills.values()).filter((s) => s.apiKeyId === apiKeyId); - } - return Array.from(this.registeredSkills.values()); + return this.scopedSkills(apiKeyId); } getSkill(identifier: string, apiKeyId?: string): Skill | undefined { - const matchesScope = (skill: Skill) => !apiKeyId || skill.apiKeyId === apiKeyId; - const skills = Array.from(this.registeredSkills.values()).filter(matchesScope); - + const skills = this.scopedSkills(apiKeyId); const byId = skills.find((skill) => skill.id === identifier); if (byId) return byId; @@ -206,8 +235,8 @@ class SkillRegistry { } getSkillVersions(name: string, apiKeyId?: string): Skill[] { - return Array.from(this.registeredSkills.values()) - .filter((skill) => skill.name === name && (!apiKeyId || skill.apiKeyId === apiKeyId)) + return this.scopedSkills(apiKeyId) + .filter((skill) => skill.name === name) .sort((a, b) => this.compareVersions(b.version, a.version)); } @@ -304,12 +333,23 @@ class SkillRegistry { try { log.debug("skills.registry.loadFromDatabase", { cached: false }); const db = getDbInstance(); - const rows = apiKeyId - ? db.prepare("SELECT * FROM skills WHERE api_key_id = ?").all(apiKeyId) - : db.prepare("SELECT * FROM skills").all(); + let rows: unknown[]; + if (!apiKeyId) { + rows = db.prepare("SELECT * FROM skills").all(); + } else if (this.isGlobalOwner(apiKeyId)) { + rows = db + .prepare("SELECT * FROM skills WHERE api_key_id IN (?, ?, ?)") + .all(...GLOBAL_SKILL_OWNER_IDS); + } else { + rows = db + .prepare("SELECT * FROM skills WHERE api_key_id IN (?, ?, ?, ?)") + .all(apiKeyId, ...GLOBAL_SKILL_OWNER_IDS); + } if (apiKeyId) { - this.removeCachedSkills((skill) => skill.apiKeyId === apiKeyId); + this.removeCachedSkills( + (skill) => skill.apiKeyId === apiKeyId || this.isGlobalOwner(skill.apiKeyId) + ); } else { this.registeredSkills.clear(); this.versionCache.clear(); diff --git a/src/lib/tokenHealthCheck.ts b/src/lib/tokenHealthCheck.ts index ffc9bcd4af2..3e38eb3dbc3 100644 --- a/src/lib/tokenHealthCheck.ts +++ b/src/lib/tokenHealthCheck.ts @@ -29,6 +29,7 @@ import { pickMaskedDisplayValue } from "@/shared/utils/maskEmail"; import { isAutomatedTestProcess } from "@/shared/utils/testProcess"; import { refreshGithubCopilotSubTokenIfNeeded } from "@/lib/tokenHealthCheckCopilot"; import { checkCursorConnectionIfNeeded } from "@/lib/tokenHealthCheckCursor"; +import { checkKimiWebConnectionIfNeeded } from "@/lib/tokenHealthCheckKimi"; const LOG_PREFIX = "[HealthCheck]"; const TRUE_ENV_VALUES = new Set(["1", "true", "yes", "on"]); @@ -603,6 +604,22 @@ export async function checkConnection(conn) { return; } + // Kimi Web proactive token check and jittered auto-refresh + const providerLower = String(conn.provider || "").toLowerCase(); + if (providerLower === "kimi-web" || providerLower === "kimi_web") { + const now = new Date().toISOString(); + await checkKimiWebConnectionIfNeeded({ + conn, + now, + log, + logWarn, + logError, + getConnectionLogLabel, + logPrefix: LOG_PREFIX, + }); + return; + } + if (!conn.refreshToken || typeof conn.refreshToken !== "string") { if (isGitHubAccessTokenOnlyConnection(conn)) { const now = new Date().toISOString(); diff --git a/src/lib/tokenHealthCheckKimi.ts b/src/lib/tokenHealthCheckKimi.ts new file mode 100644 index 00000000000..79916a14234 --- /dev/null +++ b/src/lib/tokenHealthCheckKimi.ts @@ -0,0 +1,52 @@ +import { isKimiTokenExpiringSoon } from "@omniroute/open-sse/utils/kimiJwt.ts"; +import { exchangeKimiRefreshToken } from "@/lib/kimi/tokenRefresh"; +import { updateProviderConnection } from "@/lib/db/providers"; + +export async function checkKimiWebConnectionIfNeeded(params: { + conn: any; + now: string; + log: (msg: string, ...args: any[]) => void; + logWarn: (msg: string, ...args: any[]) => void; + logError: (msg: string, ...args: any[]) => void; + getConnectionLogLabel: (conn: any) => string; + logPrefix: string; + exchangeFn?: typeof exchangeKimiRefreshToken; + persistFn?: typeof updateProviderConnection; +}): Promise { + const { conn, log, logWarn, getConnectionLogLabel, logPrefix } = params; + const provider = String(conn?.provider || "").toLowerCase(); + if (provider !== "kimi-web" && provider !== "kimi_web") return false; + + const refreshToken = conn.refreshToken || conn.providerSpecificData?.refreshToken; + if (!refreshToken) return true; // Handled, but cannot refresh without refresh_token + + const token = conn.apiKey || conn.accessToken; + // Calculate jitter: random value between 60 and 240 seconds (1 to 4 min before expiry) + const jitterSec = 60 + Math.floor(Math.random() * 180); + const expiringSoon = isKimiTokenExpiringSoon(token, jitterSec); + + if (!expiringSoon) return true; + + log(`${logPrefix} Kimi Web connection ${getConnectionLogLabel(conn)} token expiring soon; refreshing in background...`); + + const exchange = params.exchangeFn || exchangeKimiRefreshToken; + const persist = params.persistFn || updateProviderConnection; + + const res = await exchange(refreshToken); + if (res.success && res.accessToken) { + log(`${logPrefix} Kimi Web connection ${getConnectionLogLabel(conn)} token refreshed successfully.`); + await persist(conn.id, { + apiKey: res.accessToken, + accessToken: res.accessToken, + refreshToken: res.refreshToken, + expiresAt: res.expiresAtSec ? new Date(res.expiresAtSec * 1000).toISOString() : undefined, + testStatus: "active", + lastError: null, + errorCode: null, + }); + } else { + logWarn(`${logPrefix} Failed to auto-refresh Kimi Web token: ${res.error}`); + } + + return true; +} diff --git a/src/shared/components/ModelSelectField.tsx b/src/shared/components/ModelSelectField.tsx index d47fdcd77f5..a61c1b82657 100644 --- a/src/shared/components/ModelSelectField.tsx +++ b/src/shared/components/ModelSelectField.tsx @@ -4,6 +4,8 @@ import { useEffect, useState } from "react"; import { useTranslations } from "next-intl"; import Select from "./Select"; import Input from "./Input"; +import { cn } from "@/shared/utils/cn"; +import { isVisionModelId } from "@/shared/constants/visionModels"; export interface ApiModel { provider: string; @@ -29,6 +31,15 @@ export interface ModelSelectFieldProps { modelFilter?: (model: ApiModel) => boolean; /** Model API to read. The unified catalog includes specialty audio/video surfaces. */ modelSource?: "available" | "catalog"; + /** + * Render an editable text input (with a of catalog suggestions) + * instead of a plain for the known catalog AND add a + // free-text input below so a custom/unlisted id can be typed directly. The + // select keeps `label`/`allowEmpty` semantics (existing UI tests query the + // model label's parent for the onChange(e.target.value)} + options={options} + placeholder={placeholder || t("selectAModel")} + placeholderDisabled={!allowEmpty} + disabled={disabled} + aria-label={ariaLabel} + /> +
+ + onChange(e.target.value)} + className="w-full py-2 px-3 text-sm text-text-main bg-surface border border-black/10 dark:border-white/10 rounded-control focus:ring-1 focus:ring-accent/30 focus:border-accent/50 focus:outline-none transition-all disabled:opacity-50 disabled:cursor-not-allowed text-[16px] sm:text-sm" + /> +
+
+ ); + } + return (
Project⭐How it inspired OmniRoute
TOON24.9kToken-Oriented Object Notation — its columnar, header-plus-rows model shaped our tabular compaction stage.
GCF – Graph Compact Format22First inspired our tabular compaction stage; now its zero-dependency, lossless generic-profile encoder is vendored directly as the Headroom codec (MIT, SPDX-marked), current with GCF spec v3.2.
GCF – Graph Compact Format22First inspired our tabular compaction stage; now its zero-dependency, lossless generic-profile encoder is vendored directly as the Headroom codec (MIT, SPDX-marked), with later numeric-domain and count-mismatch correctness fixes.
token-optimizer-mcp444Brotli/SQLite cache + per-session context-delta — inspired our session-dedup engine.
token-savior1.1kBash-output compaction + MCP profiles — inspired our compression bail-out discipline and MCP tool-manifest reduction.
token-saver117Content-aware, per-file-type output compression with failure-aware bail-out — validated our per-type dispatch and minimum-gain skip.