From f63b4313594576d42b5f082b1945c15575400a59 Mon Sep 17 00:00:00 2001 From: Xiangzhe Date: Fri, 21 Aug 2026 15:22:10 +0800 Subject: [PATCH 001/122] fix(catalog): declare GLM reasoning effort tiers --- docs/changelog/fragments/10962.md | 1 + open-sse/config/glmProvider.ts | 29 +++++- .../config/providers/registry/zcode/index.ts | 15 ++- open-sse/executors/glm.ts | 2 +- open-sse/executors/zcode.ts | 6 +- .../__tests__/glmCodingProviderConfig.test.ts | 23 +++++ open-sse/utils/syncedEffortVariants.ts | 16 ++-- src/lib/modelMetadataRegistry.ts | 17 +++- .../glm-5.3-catalog-and-effort-tiers.test.ts | 92 ++++++++++++++++++- tests/unit/zcode-executor.test.ts | 4 +- tests/unit/zcode-provider.test.ts | 15 ++- 11 files changed, 190 insertions(+), 30 deletions(-) create mode 100644 docs/changelog/fragments/10962.md diff --git a/docs/changelog/fragments/10962.md b/docs/changelog/fragments/10962.md new file mode 100644 index 00000000000..5170a415e3f --- /dev/null +++ b/docs/changelog/fragments/10962.md @@ -0,0 +1 @@ +fix(catalog): expose only provider-routable GLM reasoning-effort tiers and remove unroutable ZCode aliases diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 9c1580e7aed..8668de2c636 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -19,17 +19,16 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ export const GLM_SHARED_MODELS = Object.freeze([ { - // GLM-5.3 (2026-08-14): one upstream id; effort is the reasoning_effort - // param (low|high|max, default max) — the -high/-low entries below are - // OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. - // Default context window not yet published by Z.ai; 1M mirrored from - // GLM-5.2 (same base model). https://z.ai/blog/glm-5.3 + // GLM-5.3 exposes low|high|max reasoning_effort (default max); -high/-low + // are OmniRoute aliases resolved by GlmExecutor::parseGlmEffortTier. + // https://docs.z.ai/guides/llm/glm-5.3 id: "glm-5.3", name: "GLM 5.3", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], }, { id: "glm-5.3-high", @@ -38,6 +37,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.3-low", @@ -46,14 +46,19 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["low"], }, { + // GLM-5.2 has two positive effective tiers: low/medium map to high and xhigh + // maps to max; disabling thinking remains the separate thinking toggle. + // https://docs.z.ai/guides/capabilities/thinking id: "glm-5.2", name: "GLM 5.2", contextLength: 1000000, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high", "max"], }, { id: "glm-5.2-high", @@ -62,6 +67,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["high"], }, { id: "glm-5.2-max", @@ -70,14 +76,18 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: ["max"], }, { + // Earlier GLM families support the thinking toggle, not reasoning_effort. + // An explicit empty list prevents generic catalog tiers from being inferred. id: "glm-5.1", name: "GLM 5.1", contextLength: 204800, maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5", @@ -86,6 +96,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-5-turbo", @@ -94,6 +105,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7-flash", @@ -102,6 +114,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.7", @@ -110,6 +123,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 131072, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.6v", @@ -118,6 +132,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -127,6 +142,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5v", @@ -135,6 +151,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], supportsVision: true, }, { @@ -144,6 +161,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, { id: "glm-4.5-air", @@ -152,6 +170,7 @@ export const GLM_SHARED_MODELS = Object.freeze([ maxOutputTokens: 32768, toolCalling: true, supportsReasoning: true, + supportedThinkingEfforts: [], }, ]); diff --git a/open-sse/config/providers/registry/zcode/index.ts b/open-sse/config/providers/registry/zcode/index.ts index cd2a4eece64..65e2e1c3d3d 100644 --- a/open-sse/config/providers/registry/zcode/index.ts +++ b/open-sse/config/providers/registry/zcode/index.ts @@ -1,6 +1,17 @@ import type { RegistryEntry } from "../../shared.ts"; import { GLM_SHARED_MODELS } from "../../../glmProvider.ts"; +const GLM_EXECUTOR_EFFORT_ALIASES = new Set([ + "glm-5.3-high", + "glm-5.3-low", + "glm-5.2-high", + "glm-5.2-max", +]); + +export const ZCODE_MODELS = GLM_SHARED_MODELS.filter( + (model) => !GLM_EXECUTOR_EFFORT_ALIASES.has(model.id) +).map((model) => ({ ...model, supportedThinkingEfforts: [] })); + /** * Local ZCode app-server backend. Authentication remains in the user's local * ZCode profile (`builtin:zai-coding-plan`); OmniRoute does not receive or @@ -14,5 +25,7 @@ export const zcodeProvider: RegistryEntry = { baseUrl: "zcode://app-server/stdio", authType: "none", authHeader: "none", - models: [...GLM_SHARED_MODELS], + // ZCode's app-server transport does not consume reasoning_effort; keep thinking + // capability metadata without advertising aliases or tiers that it would ignore. + models: ZCODE_MODELS, }; diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 6318aaab2e5..3ce6eed6c5b 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -73,7 +73,7 @@ type GlmEffortTier = { * `thinking.type=enabled` (5.3 no longer accepts thinking disabled). * * https://docs.z.ai/devpack/latest-model - * https://z.ai/blog/glm-5.3 + * https://docs.z.ai/guides/llm/glm-5.3 */ function parseGlmEffortTier(model: string): GlmEffortTier | null { switch (model) { diff --git a/open-sse/executors/zcode.ts b/open-sse/executors/zcode.ts index 0841b4daa8e..8f0a98ac142 100644 --- a/open-sse/executors/zcode.ts +++ b/open-sse/executors/zcode.ts @@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto"; import { existsSync } from "node:fs"; import { homedir } from "node:os"; import { join, resolve } from "node:path"; -import { GLM_SHARED_MODELS } from "../config/glmProvider.ts"; +import { ZCODE_MODELS } from "../config/providers/registry/zcode/index.ts"; import { BaseExecutor, type ExecuteInput, type ExecutorExecuteResult, type ProviderCredentials } from "./base.ts"; import { ZcodeAppServerClient, type ZcodeClientLike } from "./zcodeProtocol.ts"; import { buildErrorBody, errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; @@ -12,8 +12,8 @@ const DEFAULT_PROVIDER_ID = "builtin:zai-coding-plan"; const DEFAULT_TURN_TIMEOUT_MS = 120_000; const DEFAULT_POLL_INTERVAL_MS = 250; const TERMINAL_STATUSES = new Set(["completed", "idle", "paused", "error"]); -const ZCODE_MODEL_ALLOWLIST = new Set(GLM_SHARED_MODELS.map((model) => model.id)); -const DEFAULT_ZCODE_MODEL = GLM_SHARED_MODELS[0]?.id || "glm-5.2"; +const ZCODE_MODEL_ALLOWLIST = new Set(ZCODE_MODELS.map((model) => model.id)); +const DEFAULT_ZCODE_MODEL = ZCODE_MODELS[0]?.id || "glm-5.2"; type JsonRecord = Record; type OpenAIMsg = { role?: string; content?: unknown }; diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index 8b7f3cee324..50c16fa9665 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -106,6 +106,29 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("declares exact GLM reasoning-effort tiers across every shared GLM provider", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt"]) { + for (const model of getModelsByProviderId(provider)) { + expect(model.supportedThinkingEfforts, `${provider}/${model.id} effort tiers`).toEqual( + routedTiers.get(model.id) ?? [] + ); + } + } + + for (const model of getModelsByProviderId("zcode")) { + expect(model.supportedThinkingEfforts, `zcode/${model.id} effort tiers`).toEqual([]); + } + }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); diff --git a/open-sse/utils/syncedEffortVariants.ts b/open-sse/utils/syncedEffortVariants.ts index 2a2c7f0d68c..33c5ec8c56b 100644 --- a/open-sse/utils/syncedEffortVariants.ts +++ b/open-sse/utils/syncedEffortVariants.ts @@ -19,17 +19,17 @@ * only when the base model's own `supportedThinkingEfforts` actually declares that tier — * never a blind string match. * - * Skipped entirely for `codex` and `kimi`-owned models: both already own a conflicting - * native `-{effort}` suffix mechanism (`splitCodexReasoningSuffix` / - * `getKimiCodeStaticThinkingPolicy`), so double-registering here would collide with their - * own alias resolution. Also skipped for any model whose id already ends in a token that - * matches a canonical effort value, to avoid colliding with a model that legitimately ends - * in an effort-like token (e.g. a model literally named "...-high"). + * Skipped entirely for `codex`, `kimi`-owned, and GLM (`glm`, `glm-cn`, `glmt`) models: + * they already own conflicting `-{effort}` aliases (`splitCodexReasoningSuffix`, + * `getKimiCodeStaticThinkingPolicy`, or `GlmExecutor::parseGlmEffortTier`), so generating + * another layer here would create invalid nested ids. Also skipped for any model whose id + * already ends in a token that matches a canonical effort value, to avoid colliding with a + * model that legitimately ends in an effort-like token (e.g. a model named "...-high"). */ import { CANONICAL_EFFORT_VALUES } from "@/shared/reasoning/effortStandardization.ts"; -/** Provider ids that already own a native `-{effort}` suffix mechanism — never double-register. */ -export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex"]); +/** Provider ids with dedicated `-{effort}` aliases — never synthesize another suffix layer. */ +export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex", "glm", "glm-cn", "glmt"]); /** Provider-id prefixes covering that mechanism's multiple connection variants (kimi-coding, kimi-coding-apikey). */ const SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES = ["kimi"]; diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 6ed9785d73e..1c2c7e36b8e 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -130,6 +130,11 @@ function uniqueStrings(values: Array) { ]; } +export function isGlmFamilyModel(modelId: string, displayName = ""): boolean { + const glmFamilyPattern = /(?:^|[/@:_. -])glm(?=$|[-._ /@:](?:z)?\d|\d)/i; + return glmFamilyPattern.test(modelId) || glmFamilyPattern.test(displayName); +} + function toQualifiedId( providerAlias: string | null, provider: string | null, @@ -477,17 +482,19 @@ export function enrichCatalogModelEntry( // #6241: surface thinking support + the canonical effort tiers so the frontend can // render the effort/thinking toggles. `thinking` is kept for back-compat; `supportsThinking` // is the explicit flag and `effort_tiers` lists the selectable reasoning levels - // (only when the model actually supports thinking). + // (only when the model actually supports thinking). An explicit empty registry list + // is authoritative; GLM models also require a provider-declared contract instead of + // inheriting generic OpenAI effort tiers. ...(typeof metadata.capabilities.supportsThinking === "boolean" ? { thinking: metadata.capabilities.supportsThinking, supportsThinking: metadata.capabilities.supportsThinking, ...(metadata.capabilities.supportsThinking ? { - effort_tiers: - metadata.capabilities.supportedThinkingEfforts && - metadata.capabilities.supportedThinkingEfforts.length > 0 - ? [...metadata.capabilities.supportedThinkingEfforts] + effort_tiers: Array.isArray(metadata.capabilities.supportedThinkingEfforts) + ? [...metadata.capabilities.supportedThinkingEfforts] + : isGlmFamilyModel(metadata.model, metadata.displayName) + ? [] : extendCodexGpt56EffortValues( metadata.provider, metadata.model, diff --git a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts index d14c136e7b3..5d927de02a8 100644 --- a/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts +++ b/tests/unit/glm-5.3-catalog-and-effort-tiers.test.ts @@ -1,7 +1,7 @@ import test from "node:test"; import assert from "node:assert/strict"; -// GLM-5.3 support (released 2026-08-14, https://z.ai/blog/glm-5.3). +// GLM-5.3 support (released 2026-08-14, https://docs.z.ai/guides/llm/glm-5.3). // // Upstream ships ONE model id (`glm-5.3`) — effort is a request parameter // (`reasoning_effort`: low|high|max, default max) on the coding chat/completions @@ -12,14 +12,15 @@ import assert from "node:assert/strict"; // beta header), the 5.3 tiers use the documented `reasoning_effort` param on the // OpenAI coding transport. // -// Spec caveat: Z.ai has not yet published the default context window — 1M is -// mirrored from GLM-5.2 (same base model) per operator decision; correct when -// the official spec lands. +// Z.AI documents a 1M context window and 128K maximum output. -const { getRegistryEntry } = await import("../../open-sse/config/providerRegistry.ts"); +const { getRegistryEntry, REGISTRY } = await import("../../open-sse/config/providerRegistry.ts"); const { GlmExecutor } = await import("../../open-sse/executors/glm.ts"); const { MODEL_SPECS } = await import("../../src/shared/constants/modelSpecs.ts"); const { GLM_PRICING } = await import("../../src/shared/constants/pricing/shared-tiers.ts"); +const metadataRegistry = await import("../../src/lib/modelMetadataRegistry.ts"); +const { shouldExposeSyncedEffortVariants, SYNCED_EFFORT_SKIP_PROVIDERS } = + await import("../../open-sse/utils/syncedEffortVariants.ts"); const GLM_5_3_IDS = ["glm-5.3", "glm-5.3-high", "glm-5.3-low"] as const; @@ -38,6 +39,87 @@ function modelIds(provider: string): string[] { return (entry.models ?? []).map((m) => m.id); } +test("shared GLM providers keep their dedicated aliases instead of synthesizing another layer", () => { + for (const provider of ["glm", "glm-cn", "glmt"]) { + assert.ok(SYNCED_EFFORT_SKIP_PROVIDERS.has(provider), provider); + assert.equal( + shouldExposeSyncedEffortVariants({ + id: `${provider}/glm-5.3`, + owned_by: provider, + capabilities: { effort_tiers: ["low", "high", "max"] }, + }), + false, + provider + ); + } + assert.equal(SYNCED_EFFORT_SKIP_PROVIDERS.has("zcode"), false); +}); + +test("GLM family detection covers numeric, Z1, and bare provider model ids", () => { + for (const modelId of [ + "hf:zai-org/GLM-5.2", + "THUDM/GLM-Z1-32B-0414", + "THUDM/GLM-Z1-9B-0414", + "glm", + ]) { + assert.equal(metadataRegistry.isGlmFamilyModel(modelId), true, modelId); + } + assert.equal(metadataRegistry.isGlmFamilyModel("llama-3.3"), false); +}); + +test("catalog suppresses inferred tiers for every GLM registry entry without a provider contract", () => { + let audited = 0; + for (const [provider, entry] of Object.entries(REGISTRY)) { + for (const model of entry.models ?? []) { + if (!metadataRegistry.isGlmFamilyModel(model.id, model.name)) continue; + audited += 1; + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + if (capabilities.supportsThinking === true) { + assert.deepEqual( + capabilities.effort_tiers, + model.supportedThinkingEfforts ?? [], + `${provider}/${model.id}` + ); + } else { + assert.equal("effort_tiers" in capabilities, false, `${provider}/${model.id}`); + } + } + } + assert.ok(audited > 0); +}); + +test("catalog exposes only GLM effort tiers that each provider can route", () => { + const routedTiers = new Map([ + ["glm-5.3", ["low", "high", "max"]], + ["glm-5.3-high", ["high"]], + ["glm-5.3-low", ["low"]], + ["glm-5.2", ["high", "max"]], + ["glm-5.2-high", ["high"]], + ["glm-5.2-max", ["max"]], + ]); + + for (const provider of ["glm", "glm-cn", "glmt", "zcode"]) { + for (const model of getRegistryEntry(provider)!.models ?? []) { + const enriched = metadataRegistry.enrichCatalogModelEntry({ + id: `${provider}/${model.id}`, + object: "model", + owned_by: provider, + root: model.id, + }) as Record; + const capabilities = enriched.capabilities as Record; + const expected = provider === "zcode" ? [] : (routedTiers.get(model.id) ?? []); + assert.equal(capabilities.supportsThinking, true, `${provider}/${model.id}`); + assert.deepEqual(capabilities.effort_tiers, expected, `${provider}/${model.id}`); + } + } +}); + for (const provider of ["glm", "glm-cn", "glmt"]) { test(`${provider} advertises the GLM-5.3 base model and effort tiers (GLM_SHARED_MODELS)`, () => { const ids = modelIds(provider); diff --git a/tests/unit/zcode-executor.test.ts b/tests/unit/zcode-executor.test.ts index c82e7b33683..db4a8586bee 100644 --- a/tests/unit/zcode-executor.test.ts +++ b/tests/unit/zcode-executor.test.ts @@ -26,6 +26,8 @@ function requestBody() { test("ZCode accepts GLM Coding Plan models and rejects unsafe/unknown ids", async () => { const { resolveZcodeModel } = await loadZcodeExecutor(); assert.deepEqual(resolveZcodeModel("glm-5.2"), { ok: true, model: "glm-5.2" }); + assert.equal(resolveZcodeModel("glm-5.2-high").ok, false); + assert.equal(resolveZcodeModel("glm-5.3-low").ok, false); assert.equal(resolveZcodeModel("-unexpected").ok, false); assert.equal(resolveZcodeModel("unknown-model").ok, false); }); @@ -70,7 +72,7 @@ test("ZCode buffers the completed turn into OpenAI SSE when stream=true", async }); const result = await executor.execute({ - model: "glm-5.2-high", + model: "glm-5.2", body: requestBody(), stream: true, credentials: {}, diff --git a/tests/unit/zcode-provider.test.ts b/tests/unit/zcode-provider.test.ts index 3acf3c862c4..e882ab5fd91 100644 --- a/tests/unit/zcode-provider.test.ts +++ b/tests/unit/zcode-provider.test.ts @@ -10,5 +10,18 @@ test("ZCode provider registry exposes a local no-auth GLM Coding Plan backend", assert.equal(zcodeProvider.baseUrl, "zcode://app-server/stdio"); assert.equal(zcodeProvider.authType, "none"); assert.equal(zcodeProvider.authHeader, "none"); - assert.equal(zcodeProvider.models.some((model) => model.id === "glm-5.2"), true); + assert.equal( + zcodeProvider.models.some((model) => model.id === "glm-5.2"), + true + ); + for (const alias of ["glm-5.3-high", "glm-5.3-low", "glm-5.2-high", "glm-5.2-max"]) { + assert.equal( + zcodeProvider.models.some((model) => model.id === alias), + false, + alias + ); + } + for (const model of zcodeProvider.models) { + assert.deepEqual(model.supportedThinkingEfforts, [], model.id); + } }); From 410a061eafbad22305b81a06fea214f8eac99539 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 21 Aug 2026 08:02:16 -0300 Subject: [PATCH 002/122] fix(security): clear new CodeQL code-scanning alerts (round 2) (#10888) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: typecheck:core, changelog-integrity, file-size, lint todos verdes. Correção real dos 3 alertas CodeQL (HMAC em vez de hash bruto, URL parsing em vez de substring, dismiss documentado). CI vermelho é o base-red já rastreado em #9985. --- open-sse/executors/freebuff.ts | 4 +++- open-sse/services/cursorApiKeyAuth.ts | 6 +++++- tests/unit/antigravity-byop-account-rotation.test.ts | 2 +- 3 files changed, 9 insertions(+), 3 deletions(-) diff --git a/open-sse/executors/freebuff.ts b/open-sse/executors/freebuff.ts index bc2de30f632..c2cf5bdbc3c 100644 --- a/open-sse/executors/freebuff.ts +++ b/open-sse/executors/freebuff.ts @@ -1,3 +1,5 @@ +import { randomInt } from "node:crypto"; + import { BaseExecutor, type ExecuteInput, @@ -20,7 +22,7 @@ function generateClientSessionId(): string { const alphabet = "0123456789abcdefghijklmnopqrstuvwxyz"; let out = ""; for (let i = 0; i < 13; i++) { - out += alphabet[Math.floor(Math.random() * alphabet.length)]; + out += alphabet[randomInt(alphabet.length)]; } return out; } diff --git a/open-sse/services/cursorApiKeyAuth.ts b/open-sse/services/cursorApiKeyAuth.ts index 60c3385783d..126123b119d 100644 --- a/open-sse/services/cursorApiKeyAuth.ts +++ b/open-sse/services/cursorApiKeyAuth.ts @@ -50,8 +50,12 @@ export function isCursorApiKey(value: unknown): value is string { return typeof value === "string" && value.startsWith(CURSOR_API_KEY_PREFIX); } +// Session-cache key fingerprint, not a password/credential hash — keyed with a fixed context +// label so it reads as a domain-separated digest rather than a bare password hash. function cacheKeyFor(apiKey: string): string { - return crypto.createHash("sha256").update(apiKey).digest("hex"); + return crypto.createHmac("sha256", "omniroute-cursor-session-cache-fingerprint-v1") + .update(apiKey) + .digest("hex"); } export function readJwtExpiryMs(token: string): number | null { diff --git a/tests/unit/antigravity-byop-account-rotation.test.ts b/tests/unit/antigravity-byop-account-rotation.test.ts index cba1318d16b..e73b22c1847 100644 --- a/tests/unit/antigravity-byop-account-rotation.test.ts +++ b/tests/unit/antigravity-byop-account-rotation.test.ts @@ -130,7 +130,7 @@ test("Antigravity BYOP 422 rotates to a sibling account and the request succeeds { status: 200, headers: { "Content-Type": "application/json" } } ); } - if (request.url.includes("cloudcode-pa.googleapis.com")) { + if (new URL(request.url).hostname === "cloudcode-pa.googleapis.com") { modelCalls.push({ token: (request.headers.get("authorization") || "").replace(/^Bearer\s+/i, ""), }); From 0f35febd247567cdb6ddaefeda5bd12eeaca90cc Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 21 Aug 2026 08:02:19 -0300 Subject: [PATCH 003/122] fix(executors): report WS readyState in Meta AI timeout error (#10727) (#10916) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: typecheck:core, changelog-integrity, file-size, lint e 2/2 testes focados passando. Diagnóstico bem investigado do timeout WS do Meta AI (readyState exposto no erro). CI vermelho é o base-red já rastreado em #9985. --- .../10727-meta-ai-ws-timeout-diagnostics.md | 1 + open-sse/executors/muse-spark-web.ts | 2 +- ...spark-ws-timeout-diagnostics-10727.test.ts | 164 ++++++++++++++++++ 3 files changed, 166 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md create mode 100644 tests/unit/muse-spark-ws-timeout-diagnostics-10727.test.ts diff --git a/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md b/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md new file mode 100644 index 00000000000..f204684baa2 --- /dev/null +++ b/changelog.d/fixes/10727-meta-ai-ws-timeout-diagnostics.md @@ -0,0 +1 @@ +- **fix(executors):** the Meta AI (muse-spark-web) WebSocket send-message timeout now reports the socket's `readyState` at the moment it fires, so a "Meta AI WS timed out" failure can be told apart as either the connection never opening (`readyState=0`) or opening successfully and then going silent (`readyState=1`) — the exact ambiguity that made #10727 undiagnosable from logs alone (#10727). diff --git a/open-sse/executors/muse-spark-web.ts b/open-sse/executors/muse-spark-web.ts index f93191c3124..a8052c4c657 100644 --- a/open-sse/executors/muse-spark-web.ts +++ b/open-sse/executors/muse-spark-web.ts @@ -1070,7 +1070,7 @@ async function wsChat( const fail = (error: string) => finish({ content: "", deltas: [], error }); - timeout = setTimeout(() => fail("Meta AI WebSocket timed out"), 30000); + timeout = setTimeout(() => fail(`Meta AI WS timed out (readyState=${ws.readyState})`), 30000); abortHandler = () => fail("Request aborted"); signal?.addEventListener("abort", abortHandler, { once: true }); diff --git a/tests/unit/muse-spark-ws-timeout-diagnostics-10727.test.ts b/tests/unit/muse-spark-ws-timeout-diagnostics-10727.test.ts new file mode 100644 index 00000000000..98de83cce79 --- /dev/null +++ b/tests/unit/muse-spark-ws-timeout-diagnostics-10727.test.ts @@ -0,0 +1,164 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + MuseSparkWebExecutor, + __resetMuseSparkConversationCacheForTesting, + __setMuseSparkWebSocketForTesting, +} from "../../open-sse/executors/muse-spark-web.ts"; +import { WebSocket } from "ws"; + +// #10727: Meta AI (muse-spark-web) times out on the WS send-message step at +// exactly the executor's hardcoded 30s timeout, with no onerror/onclose +// firing first. The reporter's log shows the flow reaching wsChat and +// hanging until the timeout fires, meaning Meta's gateway either never +// truly opens the socket or silently drops frames after opening — but the +// old flat "Meta AI WebSocket timed out" message could not distinguish +// those two failure modes for whoever debugs the next occurrence. +// +// Root-causing (and fixing) the reverse-engineered private WS protocol +// itself requires a live meta.ai session + a fresh DevTools capture (see +// the plan-file's `needs-vps` verdict) — not achievable in this sandbox. +// This regression test locks in the diagnosability improvement that *is* +// verifiable here: the timeout error now reports the socket's readyState +// at the moment it fires, so a future report can tell "never opened" +// (readyState 0) apart from "opened but Meta went silent" (readyState 1). + +/** + * Intercepts only the wsChat 30000ms timeout registration and lets every + * other setTimeout (including the mock WebSocket's own onopen scheduling) + * run for real. Firing the captured callback directly — instead of + * advancing a fake clock — keeps the test fast and avoids interleaving + * bugs between fake timers and the executor's real async/await chain. + */ +function interceptWsTimeout(): { fire: () => void; restore: () => void } { + const original = globalThis.setTimeout; + let captured: (() => void) | null = null; + globalThis.setTimeout = ((cb: (...a: unknown[]) => void, ms?: number, ...args: unknown[]) => { + if (ms === 30000 && captured === null) { + captured = cb as () => void; + return 0 as unknown as ReturnType; + } + return original(cb as () => void, ms, ...args); + }) as typeof setTimeout; + return { + fire: () => { + assert.ok(captured, "the 30000ms wsChat timeout was never registered"); + captured?.(); + }, + restore: () => { + globalThis.setTimeout = original; + }, + }; +} + +class NeverOpensWebSocket { + onopen: (() => void) | null = null; + onmessage: ((evt: { data: string }) => void) | null = null; + onclose: (() => void) | null = null; + onerror: ((evt: Error) => void) | null = null; + readyState = WebSocket.CONNECTING; + url: string; + constructor(url: string) { + this.url = url; + // Never calls onopen, onmessage, onerror, or onclose — mirrors the + // reported symptom exactly: the socket just hangs until the timeout. + } + send(_data: Uint8Array | string) {} + close() {} +} + +class OpensThenSilentWebSocket { + onopen: (() => void) | null = null; + onmessage: ((evt: { data: string }) => void) | null = null; + onclose: (() => void) | null = null; + onerror: ((evt: Error) => void) | null = null; + readyState = WebSocket.CONNECTING; + url: string; + constructor(url: string) { + this.url = url; + setTimeout(() => { + this.readyState = WebSocket.OPEN; + this.onopen?.(); + }, 0); + } + send(_data: Uint8Array | string) {} + close() {} +} + +function baseInput(connectionId: string): Parameters[0] { + return { + model: "muse-spark", + body: { messages: [{ role: "user", content: "ping" }] }, + stream: false, + credentials: { + apiKey: "ecto_1_sess=test123", + connectionId, + providerSpecificData: { authorization: "ecto1:test-auth-token" }, + }, + signal: null, + log: null, + upstreamExtraHeaders: undefined, + } as Parameters[0]; +} + +test("#10727: WS timeout while still CONNECTING reports readyState=0 (never opened)", async () => { + __resetMuseSparkConversationCacheForTesting(); + const executor = new MuseSparkWebExecutor(); + const originalFetch = globalThis.fetch; + globalThis.fetch = async () => new Response("{}", { status: 200 }); + const restore = __setMuseSparkWebSocketForTesting( + NeverOpensWebSocket as unknown as typeof WebSocket + ); + const timeoutHook = interceptWsTimeout(); + try { + const resultPromise = executor.execute(baseInput("conn-10727-never-opens")); + // Let the GraphQL warmup/mode-switch awaits and the WS constructor run + // before the 30s timeout is registered. + await new Promise((r) => setTimeout(r, 20)); + timeoutHook.fire(); + + const result = await resultPromise; + assert.equal(result.response.status, 502); + const body = await result.response.json(); + assert.match( + body.error.message, + /readyState=0/, + "timeout while the socket never left CONNECTING must report readyState=0" + ); + } finally { + globalThis.fetch = originalFetch; + restore(); + timeoutHook.restore(); + } +}); + +test("#10727: WS timeout after a successful open reports readyState=1 (opened, then silent)", async () => { + __resetMuseSparkConversationCacheForTesting(); + const executor = new MuseSparkWebExecutor(); + const originalFetch = globalThis.fetch; + globalThis.fetch = async () => new Response("{}", { status: 200 }); + const restore = __setMuseSparkWebSocketForTesting( + OpensThenSilentWebSocket as unknown as typeof WebSocket + ); + const timeoutHook = interceptWsTimeout(); + try { + const resultPromise = executor.execute(baseInput("conn-10727-opens-silent")); + // Let the GraphQL awaits run, the WS open (its own real setTimeout(...,0)), + // and the intro/prompt frames send before the 30s timeout is registered. + await new Promise((r) => setTimeout(r, 20)); + timeoutHook.fire(); + + const result = await resultPromise; + assert.equal(result.response.status, 502); + const body = await result.response.json(); + assert.match( + body.error.message, + /readyState=1/, + "timeout after the socket reached OPEN must report readyState=1, not the never-opened case" + ); + } finally { + globalThis.fetch = originalFetch; + restore(); + timeoutHook.restore(); + } +}); From a8ca9575f2161652926c91f0f121e18539aafd08 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 21 Aug 2026 08:02:22 -0300 Subject: [PATCH 004/122] fix: add cold-restart native-driver regression check to electron smoke (#7592) (#10921) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: typecheck:core, changelog-integrity, file-size, lint e 7/7 testes focados passando. Investigação completa com verificação de ancestralidade via merge-base antes de fechar a issue original. CI vermelho é o base-red já rastreado em #9985. --- .github/workflows/electron-release.yml | 5 + ...ectron-cold-restart-native-driver-check.md | 1 + scripts/dev/smoke-electron-packaged.mjs | 209 ++++++++++++------ tests/unit/electron-smoke-script.test.ts | 26 +++ 4 files changed, 169 insertions(+), 72 deletions(-) create mode 100644 changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml index 33708fc426a..e899a664eaa 100644 --- a/.github/workflows/electron-release.yml +++ b/.github/workflows/electron-release.yml @@ -279,9 +279,14 @@ jobs: - name: Smoke packaged Electron app (Linux) if: matrix.platform == 'linux' + # #7592: also cold-restart against the same DATA_DIR and assert a + # native SQLite driver (not the sql.js WASM fallback) is selected on + # the second launch — blocking here since Linux has no Windows-style + # sandbox caveats that would make it flaky. env: ELECTRON_SMOKE_TIMEOUT_MS: 60000 ELECTRON_SMOKE_STREAM_LOGS: "1" + ELECTRON_SMOKE_COLD_RESTART: "1" run: xvfb-run -a npm run electron:smoke:packaged - name: Collect installers diff --git a/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md b/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md new file mode 100644 index 00000000000..e6458879cf0 --- /dev/null +++ b/changelog.d/fixes/7592-electron-cold-restart-native-driver-check.md @@ -0,0 +1 @@ +- **Electron packaged smoke test:** add a cold-restart mode (`ELECTRON_SMOKE_COLD_RESTART=1`, wired blocking on the Linux release leg) that relaunches the packaged app against its own persisted `DATA_DIR` and asserts a native SQLite driver was selected instead of the sql.js WASM fallback, closing the regression-test gap flagged in the stale-ABI `better-sqlite3` investigation ([#7592](https://github.com/diegosouzapw/OmniRoute/issues/7592)). diff --git a/scripts/dev/smoke-electron-packaged.mjs b/scripts/dev/smoke-electron-packaged.mjs index 03717912ff7..72afc2f4a7c 100644 --- a/scripts/dev/smoke-electron-packaged.mjs +++ b/scripts/dev/smoke-electron-packaged.mjs @@ -409,45 +409,115 @@ async function settleAfterReady({ getExitState, logs, settleMs }) { } } -async function main() { - const appExecutable = discoverPackagedExecutable(); - if (!existsSync(appExecutable)) { +function assertExecutableExists(appExecutable) { + if (existsSync(appExecutable)) return; + + throw new Error( + `Packaged OmniRoute executable not found at ${appExecutable}. Build it first with \`npm run build: --prefix electron\` or set ELECTRON_SMOKE_APP_EXECUTABLE.` + ); +} + +// ── CI sandbox workaround ────────────────────────────────── +// GitHub Actions runners cannot set SUID on chrome-sandbox (Linux) +// and Windows runners may fail silently without --no-sandbox. +function buildCiSpawnArgs(currentPlatform = platform()) { + if (!process.env.CI) return []; + + const spawnArgs = ["--no-sandbox", "--disable-gpu"]; + if (currentPlatform === "linux") { + spawnArgs.push("--disable-dev-shm-usage"); + } + return spawnArgs; +} + +const NATIVE_DRIVER_LOG_PATTERN = /\[DB\] Driver: (bun:sqlite|better-sqlite3|node:sqlite) \|/; +const SQLJS_DRIVER_LOG_PATTERN = /\[DB\] Driver: sql\.js \|/; + +/** + * Regression guard for #7592: on a packaged app's SECOND launch against an + * already-persisted DATA_DIR, a stale-ABI better-sqlite3 binary (resolved via + * a Turbopack-hashed import) used to fail to load and silently fall through + * to the sql.js (WASM) driver — which then OOMs/retry-loops on real-sized + * databases. Asserts the startup log shows a native driver was selected. + */ +export function assertNativeDriverSelected(logs) { + if (NATIVE_DRIVER_LOG_PATTERN.test(logs)) return; + + if (SQLJS_DRIVER_LOG_PATTERN.test(logs)) { throw new Error( - `Packaged OmniRoute executable not found at ${appExecutable}. Build it first with \`npm run build: --prefix electron\` or set ELECTRON_SMOKE_APP_EXECUTABLE.` + "Packaged Electron app fell back to the sql.js (WASM) driver instead of a native SQLite " + + "driver — this is the regression #7592 guards against (stale-ABI better-sqlite3 binary)." ); } - const smokeUrl = process.env.ELECTRON_SMOKE_URL || DEFAULT_URL; - const timeoutMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_TIMEOUT_MS, DEFAULT_TIMEOUT_MS); - const settleMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_SETTLE_MS, DEFAULT_SETTLE_MS); - const dataDir = - process.env.ELECTRON_SMOKE_DATA_DIR || - (await mkdtemp(join(tmpdir(), "omniroute-electron-smoke-"))); - const removeDataDir = - !process.env.ELECTRON_SMOKE_DATA_DIR && process.env.ELECTRON_SMOKE_KEEP_DATA !== "1"; - const smokeEnv = buildSmokeEnv({ dataDir }); + throw new Error( + "Packaged Electron app logs contain no '[DB] Driver: ...' line — cannot confirm which SQLite " + + "driver loaded." + ); +} - await assertPortIsFree(smokeUrl); - await ensureSmokeEnvDirs(smokeEnv, dataDir); +async function waitForReady({ logs, smokeUrl, timeoutMs, settleMs, exitState }) { + const startedAt = Date.now(); + let lastError = null; - // ── CI sandbox workaround ────────────────────────────────── - // GitHub Actions runners cannot set SUID on chrome-sandbox (Linux) - // and Windows runners may fail silently without --no-sandbox. - const spawnArgs = []; - if (process.env.CI) { - spawnArgs.push("--no-sandbox", "--disable-gpu"); - if (platform() === "linux") { - spawnArgs.push("--disable-dev-shm-usage"); + while (Date.now() - startedAt < timeoutMs) { + assertNoFatalLogs(logs.value); + + if (exitState.spawnError !== null) { + throw new Error(`Packaged Electron app failed to launch: ${exitState.spawnError.message}`); } + if (exitState.exitCode !== null || exitState.signalCode !== null) { + throw new Error( + `Packaged Electron app exited before readiness: code=${exitState.exitCode} signal=${exitState.signalCode}` + ); + } + + try { + const response = await fetchWithTimeout(smokeUrl, 1_000); + if (response.status === 200) { + assertNoFatalLogs(logs.value); + console.log(`[electron-smoke] ready: ${smokeUrl} returned HTTP 200`); + await settleAfterReady({ + getExitState: () => ({ exitCode: exitState.exitCode, signalCode: exitState.signalCode }), + logs, + settleMs, + }); + console.log(`[electron-smoke] stable for ${settleMs}ms after readiness`); + return; + } + lastError = new Error(`HTTP ${response.status}`); + } catch (error) { + lastError = error; + } + + await sleep(500); } + throw new Error( + `Packaged Electron app did not serve ${smokeUrl} within ${timeoutMs}ms. Last error: ${ + lastError instanceof Error ? lastError.message : String(lastError) + }` + ); +} + +/** + * Launches the packaged app once against `dataDir`, waits for readiness + + * settle, tears it down, and returns the captured stdout/stderr text. Shared + * by the single-launch path and the cold-restart (two-launch) path so both + * exercise identical spawn/readiness/shutdown behavior. + */ +async function launchAndCollectLogs({ appExecutable, smokeUrl, dataDir, timeoutMs, settleMs, streamLogs }) { + const smokeEnv = buildSmokeEnv({ dataDir }); + await assertPortIsFree(smokeUrl); + await ensureSmokeEnvDirs(smokeEnv, dataDir); + + const spawnArgs = buildCiSpawnArgs(); console.log(`[electron-smoke] launching ${appExecutable}`); if (spawnArgs.length) console.log(`[electron-smoke] CI args: ${spawnArgs.join(" ")}`); console.log(`[electron-smoke] DATA_DIR=${dataDir}`); console.log(`[electron-smoke] waiting for ${smokeUrl}`); const logs = { value: "" }; - const streamLogs = process.env.ELECTRON_SMOKE_STREAM_LOGS === "1"; const child = spawn(appExecutable, spawnArgs, { detached: platform() !== "win32", env: smokeEnv, @@ -457,60 +527,18 @@ async function main() { child.stdout?.on("data", (chunk) => appendLog(logs, chunk, "[electron] ", streamLogs)); child.stderr?.on("data", (chunk) => appendLog(logs, chunk, "[electron:err] ", streamLogs)); - let exitCode = null; - let signalCode = null; - let spawnError = null; + const exitState = { exitCode: null, signalCode: null, spawnError: null }; child.once("exit", (code, signal) => { - exitCode = code; - signalCode = signal; + exitState.exitCode = code; + exitState.signalCode = signal; }); child.once("error", (error) => { - spawnError = error; + exitState.spawnError = error; }); try { - const startedAt = Date.now(); - let lastError = null; - - while (Date.now() - startedAt < timeoutMs) { - assertNoFatalLogs(logs.value); - - if (spawnError !== null) { - throw new Error(`Packaged Electron app failed to launch: ${spawnError.message}`); - } - - if (exitCode !== null || signalCode !== null) { - throw new Error( - `Packaged Electron app exited before readiness: code=${exitCode} signal=${signalCode}` - ); - } - - try { - const response = await fetchWithTimeout(smokeUrl, 1_000); - if (response.status === 200) { - assertNoFatalLogs(logs.value); - console.log(`[electron-smoke] ready: ${smokeUrl} returned HTTP 200`); - await settleAfterReady({ - getExitState: () => ({ exitCode, signalCode }), - logs, - settleMs, - }); - console.log(`[electron-smoke] stable for ${settleMs}ms after readiness`); - return; - } - lastError = new Error(`HTTP ${response.status}`); - } catch (error) { - lastError = error; - } - - await new Promise((resolve) => setTimeout(resolve, 500)); - } - - throw new Error( - `Packaged Electron app did not serve ${smokeUrl} within ${timeoutMs}ms. Last error: ${ - lastError instanceof Error ? lastError.message : String(lastError) - }` - ); + await waitForReady({ logs, smokeUrl, timeoutMs, settleMs, exitState }); + return logs.value; } catch (error) { if (!streamLogs) { printLogTail(logs.value); @@ -519,6 +547,43 @@ async function main() { } finally { await stopApp(child); await waitForPortClosed(smokeUrl); + } +} + +async function main() { + const appExecutable = discoverPackagedExecutable(); + assertExecutableExists(appExecutable); + + const smokeUrl = process.env.ELECTRON_SMOKE_URL || DEFAULT_URL; + const timeoutMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_TIMEOUT_MS, DEFAULT_TIMEOUT_MS); + const settleMs = parsePositiveInteger(process.env.ELECTRON_SMOKE_SETTLE_MS, DEFAULT_SETTLE_MS); + const streamLogs = process.env.ELECTRON_SMOKE_STREAM_LOGS === "1"; + // #7592: rerun against the SAME (persisted) DATA_DIR and assert the second + // launch selected a native SQLite driver, not the sql.js WASM fallback. + const coldRestart = process.env.ELECTRON_SMOKE_COLD_RESTART === "1"; + const dataDir = + process.env.ELECTRON_SMOKE_DATA_DIR || + (await mkdtemp(join(tmpdir(), "omniroute-electron-smoke-"))); + const removeDataDir = + !process.env.ELECTRON_SMOKE_DATA_DIR && process.env.ELECTRON_SMOKE_KEEP_DATA !== "1"; + + try { + await launchAndCollectLogs({ appExecutable, smokeUrl, dataDir, timeoutMs, settleMs, streamLogs }); + + if (!coldRestart) return; + + console.log("[electron-smoke] cold-restart: relaunching against the same DATA_DIR"); + const secondLaunchLogs = await launchAndCollectLogs({ + appExecutable, + smokeUrl, + dataDir, + timeoutMs, + settleMs, + streamLogs, + }); + assertNativeDriverSelected(secondLaunchLogs); + console.log("[electron-smoke] cold-restart: native SQLite driver confirmed on second launch"); + } finally { if (removeDataDir) { await rm(dataDir, { recursive: true, force: true }); } diff --git a/tests/unit/electron-smoke-script.test.ts b/tests/unit/electron-smoke-script.test.ts index 9f3bea9f8e4..f7167859e16 100644 --- a/tests/unit/electron-smoke-script.test.ts +++ b/tests/unit/electron-smoke-script.test.ts @@ -2,6 +2,7 @@ import assert from "node:assert/strict"; import test from "node:test"; import { + assertNativeDriverSelected, buildSmokeEnv, FATAL_LOG_PATTERNS, LINUX_EXECUTABLE_NAMES, @@ -71,3 +72,28 @@ test("electron smoke force-terminates the Windows process tree before the parent assert.deepEqual(signals, ["SIGKILL"]); assert.deepEqual(waits, [2_000]); }); + +// #7592: on a cold restart against an already-persisted DATA_DIR, a stale-ABI +// better-sqlite3 binary used to fail to load and silently fall through to the +// sql.js (WASM) driver. These are the regression guards for that assertion. +test("electron smoke accepts every native SQLite driver on the startup log", () => { + for (const driver of ["bun:sqlite", "better-sqlite3", "node:sqlite"]) { + assert.doesNotThrow(() => + assertNativeDriverSelected(`[electron] [DB] Driver: ${driver} | file: /data/storage.sqlite`) + ); + } +}); + +test("electron smoke flags a cold-restart fallback to the sql.js WASM driver", () => { + assert.throws( + () => assertNativeDriverSelected("[electron] [DB] Driver: sql.js | file: /data/storage.sqlite"), + /fell back to the sql\.js \(WASM\) driver/ + ); +}); + +test("electron smoke flags startup logs missing any driver selection line", () => { + assert.throws( + () => assertNativeDriverSelected("[electron] [server] listening on 20128"), + /no '\[DB\] Driver: \.\.\.' line/ + ); +}); From 7011c5fafaabf00980cd3ad6dd4f912b95db56f1 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 21 Aug 2026 08:02:26 -0300 Subject: [PATCH 005/122] fix(cli): repair hollow externalized package dirs in nested distDir node_modules (#7346) (#10924) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: typecheck:core, changelog-integrity, file-size, lint e teste focado passando. Root cause bem documentado (distDir customizado gera dois node_modules externalizados). CI vermelho é o base-red já rastreado em #9985. --- ...6-electron-hollow-nested-package-repair.md | 1 + scripts/build/assembleStandalone.mjs | 28 +++++++--- ...empty-external-package-dirs-nested.test.ts | 55 +++++++++++++++++++ 3 files changed, 75 insertions(+), 9 deletions(-) create mode 100644 changelog.d/fixes/7346-electron-hollow-nested-package-repair.md create mode 100644 tests/unit/build/repair-empty-external-package-dirs-nested.test.ts diff --git a/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md b/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md new file mode 100644 index 00000000000..fd7e61988ba --- /dev/null +++ b/changelog.d/fixes/7346-electron-hollow-nested-package-repair.md @@ -0,0 +1 @@ +- fix(cli): repair hollow externalized package dirs in the nested `/node_modules` bundle location too, not just the top-level one, fixing macOS/Linux Electron `ERR_MODULE_NOT_FOUND` on Turbopack-externalized packages (#7346) diff --git a/scripts/build/assembleStandalone.mjs b/scripts/build/assembleStandalone.mjs index b4c8d12c1fa..ee8d730ccf2 100644 --- a/scripts/build/assembleStandalone.mjs +++ b/scripts/build/assembleStandalone.mjs @@ -628,12 +628,11 @@ function copyNativeAssetsAndExtraModules(projectRoot, resolvedOutDir) { * This keeps the fix narrowly scoped to packages the standalone already expects. * * @param {string} projectRoot - * @param {string} resolvedOutDir + * @param {string} bundleNodeModules * @returns {{repaired: number, packages: string[]}} */ -function repairEmptyExternalPackageDirs(projectRoot, resolvedOutDir) { +function repairEmptyExternalPackageDirs(projectRoot, bundleNodeModules) { const summary = { repaired: 0, packages: [] }; - const bundleNodeModules = path.join(resolvedOutDir, "node_modules"); const sourceNodeModules = path.join(projectRoot, "node_modules"); if (!fsSync.existsSync(bundleNodeModules) || !fsSync.existsSync(sourceNodeModules)) { return summary; @@ -899,12 +898,23 @@ export function assembleStandalone({ // 6. Optionally copy native assets + extra modules (synchronous) if (copyNatives) { copyNativeAssetsAndExtraModules(projectRoot, resolvedOutDir); - const emptyPkgRepair = repairEmptyExternalPackageDirs(projectRoot, resolvedOutDir); - if (emptyPkgRepair.repaired > 0) { - console.log( - `[assembleStandalone] Repaired ${emptyPkgRepair.repaired} hollow external package dir(s): ` + - emptyPkgRepair.packages.join(", ") - ); + // Repair hollow externalized package dirs in BOTH locations Turbopack's standalone + // tracer can populate: the top-level bundle node_modules, and — for projects with a + // custom distDir (see next.config.mjs) — the nested /node_modules mirrored + // alongside the traced server chunks. materializeBundledSymlinks (step 7 below) already + // treats these as two distinct targets; #9913 only covered the top-level one, which left + // the nested location's hollow dirs unrepaired (#7346). + for (const bundleNodeModules of [ + path.join(resolvedOutDir, "node_modules"), + path.join(resolvedOutDir, relDistDir, "node_modules"), + ]) { + const emptyPkgRepair = repairEmptyExternalPackageDirs(projectRoot, bundleNodeModules); + if (emptyPkgRepair.repaired > 0) { + console.log( + `[assembleStandalone] Repaired ${emptyPkgRepair.repaired} hollow external package dir(s) in ` + + `${path.relative(resolvedOutDir, bundleNodeModules) || "."}: ${emptyPkgRepair.packages.join(", ")}` + ); + } } // #9166: dynamically imported LLMLingua packages are not reliably traced diff --git a/tests/unit/build/repair-empty-external-package-dirs-nested.test.ts b/tests/unit/build/repair-empty-external-package-dirs-nested.test.ts new file mode 100644 index 00000000000..50af15cadce --- /dev/null +++ b/tests/unit/build/repair-empty-external-package-dirs-nested.test.ts @@ -0,0 +1,55 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +import { assembleStandalone } from "../../../scripts/build/assembleStandalone.mjs"; + +// #7346: on macOS (and Linux AppImage) Electron builds, Turbopack's standalone tracer can leave +// a hollow (directory exists, contains zero files) externalized-package directory behind. #9913 +// added `repairEmptyExternalPackageDirs` to overlay the real source package on top of a hollow +// bundle dir — but it only scans the TOP-LEVEL `/node_modules`. This project builds with +// a custom, non-default `distDir` (".build/next", see next.config.mjs), and Next's standalone +// tracer also emits a SECOND, nested `node_modules` under `//node_modules` +// (the same location `materializeBundledSymlinks` already treats as a distinct target — see +// assembleStandalone() step 7). A hollow externalized package dir landing in that nested +// location is never repaired, which reproduces the exact ERR_MODULE_NOT_FOUND class reported on +// #7346 even after #6794/#7353/#9913 all landed. +test("assembleStandalone repairs a hollow externalized package dir in the nested node_modules, not just the top-level one", () => { + const tmp = fs.mkdtempSync(path.join(os.tmpdir(), "repair-nested-empty-pkg-")); + const projectRoot = path.join(tmp, "project"); + const relDistDir = ".build/next"; + const distDir = path.join(projectRoot, relDistDir); + const outDir = path.join(tmp, "dist"); + + // Real source package the repair should copy from. + const sourcePkgDir = path.join(projectRoot, "node_modules", "some-nested-pkg"); + fs.mkdirSync(sourcePkgDir, { recursive: true }); + fs.writeFileSync(path.join(sourcePkgDir, "package.json"), '{"name":"some-nested-pkg"}'); + fs.writeFileSync(path.join(sourcePkgDir, "index.js"), "module.exports = {};"); + + // Fake standalone tree with a hollow externalized package dir under the NESTED + // /node_modules (directory exists but contains zero files — the exact + // "hollow" shape repairEmptyExternalPackageDirs already repairs at the top level). + const standaloneDir = path.join(distDir, "standalone"); + fs.mkdirSync(standaloneDir, { recursive: true }); + fs.writeFileSync(path.join(standaloneDir, "server.js"), "// server"); + const hollowNestedPkgDir = path.join(standaloneDir, relDistDir, "node_modules", "some-nested-pkg"); + fs.mkdirSync(hollowNestedPkgDir, { recursive: true }); + + assembleStandalone({ + distDir, + outDir, + projectRoot, + copyNatives: true, + }); + + const repairedIndexPath = path.join(outDir, relDistDir, "node_modules", "some-nested-pkg", "index.js"); + assert.ok( + fs.existsSync(repairedIndexPath), + "hollow nested externalized package dir must be repaired with the real source package (index.js present)" + ); + + fs.rmSync(tmp, { recursive: true, force: true }); +}); From 14d2c90e231412c22f215076c03cd003d77e87f6 Mon Sep 17 00:00:00 2001 From: Jack Smith Date: Fri, 21 Aug 2026 19:25:03 +0800 Subject: [PATCH 006/122] fix(executor): respect apiType="chat" in forceResponsesUpstream (#5483 regression) (#10946) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: typecheck:core, changelog-integrity, complexity, cognitive-complexity, file-size, lint e testes focados todos verdes. Regressão real corrigida (apiType=chat agora é honrado em vez de forçado para /responses). CI vermelho é o base-red já rastreado em #9985. --- open-sse/executors/forceResponsesUpstream.ts | 11 +++++ tests/unit/executor-default-base.test.ts | 51 ++++++++++++++++++++ 2 files changed, 62 insertions(+) diff --git a/open-sse/executors/forceResponsesUpstream.ts b/open-sse/executors/forceResponsesUpstream.ts index 0c545d980e6..4de8961a3b5 100644 --- a/open-sse/executors/forceResponsesUpstream.ts +++ b/open-sse/executors/forceResponsesUpstream.ts @@ -28,6 +28,17 @@ export function shouldForceResponsesUpstream( const providerSpecificData = credentials?.providerSpecificData ?? null; if (providerSpecificData?._omnirouteForceResponsesUpstream === true) return true; if (getOpenAICompatibleType(provider, providerSpecificData) === "responses") return false; + // apiType="chat" means the operator explicitly chose the chat/completions + // wire. Don't second-guess that choice by forcing /responses just because the + // body carries namespace tools — the standard namespace→flatten path + // (openai-responses.ts) handles those correctly for chat backends. + if ( + providerSpecificData && + typeof providerSpecificData.apiType === "string" && + providerSpecificData.apiType === "chat" + ) { + return false; + } const hasResponsesShape = body.input !== undefined || diff --git a/tests/unit/executor-default-base.test.ts b/tests/unit/executor-default-base.test.ts index f144d7f2242..0e216780541 100644 --- a/tests/unit/executor-default-base.test.ts +++ b/tests/unit/executor-default-base.test.ts @@ -9,6 +9,7 @@ import { mergeUpstreamExtraHeaders, setUserAgentHeader, } from "../../open-sse/executors/base.ts"; +import { shouldForceResponsesUpstream } from "../../open-sse/executors/forceResponsesUpstream.ts"; import { DefaultExecutor } from "../../open-sse/executors/default.ts"; import { PROVIDERS } from "../../open-sse/config/constants.ts"; import { @@ -1578,3 +1579,53 @@ test("DefaultExecutor.execute does not produce duplicate anthropic-version heade /^x-anthropic-billing-header: cc_version=2\.1\.220\.1f2; cc_entrypoint=cli; cch=[0-9a-f]{5};$/ ); }); + +test('shouldForceResponsesUpstream respects explicit apiType="chat" even when namespace tools are present', () => { + const body = { + input: "hi", + tools: [ + { + type: "namespace", + name: "collaboration", + tools: [ + { + name: "spawn_agent", + description: "Spawn an agent", + parameters: { type: "object", properties: { task: { type: "string" } } }, + }, + ], + }, + ], + }; + const credentials = { + providerSpecificData: { + baseUrl: "https://ark.cn-beijing.volces.com/api/coding/v3", + apiType: "chat", + }, + }; + assert.equal( + shouldForceResponsesUpstream("openai-compatible-responses-demo", body, credentials), + false + ); +}); + +test("shouldForceResponsesUpstream still forces /responses for untyped OpenAI-compatible providers with namespace tools", () => { + const body = { + input: "hi", + tools: [ + { + type: "namespace", + name: "collaboration", + tools: [ + { name: "spawn_agent", description: "Spawn an agent", parameters: { type: "object" } }, + ], + }, + ], + }; + const credentials = { + providerSpecificData: { + baseUrl: "https://proxy.example/v1", + }, + }; + assert.equal(shouldForceResponsesUpstream("openai-compatible-test", body, credentials), true); +}); From e708030adff1f440aa96747ca5e620f7d94414cd Mon Sep 17 00:00:00 2001 From: Nguyen Thanh Dat Date: Fri, 21 Aug 2026 18:25:07 +0700 Subject: [PATCH 007/122] fix(resilience): make least-used rotate by recording the use it sorts on (#10945) (#10951) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: mesmos gates + testes focados verdes. Bug real e bem reproduzido (least-used nunca gravava lastUsedAt, sempre a mesma conexão escolhida). CI vermelho é o base-red já rastreado em #9985. --- .../fixes/10945-least-used-rotation.md | 1 + src/sse/services/auth.ts | 9 ++ tests/unit/least-used-rotation-10945.test.ts | 140 ++++++++++++++++++ 3 files changed, 150 insertions(+) create mode 100644 changelog.d/fixes/10945-least-used-rotation.md create mode 100644 tests/unit/least-used-rotation-10945.test.ts diff --git a/changelog.d/fixes/10945-least-used-rotation.md b/changelog.d/fixes/10945-least-used-rotation.md new file mode 100644 index 00000000000..36b23951b25 --- /dev/null +++ b/changelog.d/fixes/10945-least-used-rotation.md @@ -0,0 +1 @@ +- **Account rotation:** make `fallbackStrategy: "least-used"` actually rotate. The strategy sorts on `lastUsedAt` but never wrote it — only the round-robin branch committed — so on a pool where every `last_used_at` was still `NULL` the tie-break fell through to `priority` and returned the same connection on every dispatch ([#10945](https://github.com/diegosouzapw/OmniRoute/issues/10945)). diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index 8f456f5566f..822a92fd562 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -1967,6 +1967,15 @@ export async function getProviderCredentials( return new Date(a.lastUsedAt).getTime() - new Date(b.lastUsedAt).getTime(); }); connection = sorted[0]; + // Record the use (#10945). This strategy sorts on the very field it was + // not writing, so on a pool where every lastUsedAt is null the tie-break + // fell through to `priority` and returned the SAME connection on every + // call, forever — the opposite of the documented behaviour, and silent. + // round-robin is the only other strategy that reads lastUsedAt and it has + // always committed here; least-used now does the same. + const commit = planLastUsedCommit(connection, connectionsRaw, 1); + if (options.lease) commitSelectionSideEffects = commit; + else await commit(); } else if (strategy === "cost-optimized") { // Cost Optimized: sort by priority ascending (lower = cheaper/preferred) // Future: can be enhanced with actual cost data per provider diff --git a/tests/unit/least-used-rotation-10945.test.ts b/tests/unit/least-used-rotation-10945.test.ts new file mode 100644 index 00000000000..6218697ce1b --- /dev/null +++ b/tests/unit/least-used-rotation-10945.test.ts @@ -0,0 +1,140 @@ +// #10945 — `fallbackStrategy: "least-used"` never rotated on a fresh pool. +// +// The strategy sorts on `lastUsedAt` but was the only lastUsedAt-reading +// strategy that never wrote it: `planLastUsedCommit()` was called from the +// round-robin branch only. With every row's `last_used_at` still NULL the +// tie-break fell through to `priority` ascending, so the same connection came +// back on every dispatch — silently, with no error or warning. +// +// Measured on the pre-fix build, three sequential dispatches over a 3-account +// pool: +// +// least-used picked [0, 0, 0] last_used_at [null, null, null] +// round-robin picked [0, 1, 2] last_used_at [ts, ts, ts ] +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-least-used-10945-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const settingsDb = await import("../../src/lib/db/settings.ts"); +const auth = await import("../../src/sse/services/auth.ts"); + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +/** Three active apikey connections, distinct priorities, all last_used_at NULL. */ +async function seedFreshPool(): Promise { + const ids: string[] = []; + for (const [index, apiKey] of ["sk-lu-a", "sk-lu-b", "sk-lu-c"].entries()) { + const connection = (await providersDb.createProviderConnection({ + provider: "glm", + authType: "apikey", + apiKey, + isActive: true, + testStatus: "active", + priority: index + 1, + })) as { id: string }; + ids.push(connection.id); + } + return ids; +} + +test("#10945 least-used rotates across a fresh pool instead of pinning one account", async () => { + await resetStorage(); + const ids = await seedFreshPool(); + await settingsDb.updateSettings({ fallbackStrategy: "least-used" }); + + const picked: string[] = []; + for (let i = 0; i < ids.length; i++) { + const selected = (await auth.getProviderCredentials("glm", null, null, "glm-4.6")) as { + connectionId: string; + } | null; + assert.ok(selected, `dispatch ${i + 1} should resolve a connection`); + picked.push(selected.connectionId); + } + + assert.equal( + new Set(picked).size, + ids.length, + `least-used must visit every account before repeating; got ${JSON.stringify( + picked.map((id) => ids.indexOf(id)) + )}` + ); +}); + +test("#10945 least-used persists last_used_at so the next process keeps rotating", async () => { + await resetStorage(); + const ids = await seedFreshPool(); + await settingsDb.updateSettings({ fallbackStrategy: "least-used" }); + + for (let i = 0; i < ids.length; i++) { + await auth.getProviderCredentials("glm", null, null, "glm-4.6"); + } + + const rows = await Promise.all( + ids.map((id) => providersDb.getProviderConnectionById(id) as Promise<{ lastUsedAt?: string }>) + ); + for (const [index, row] of rows.entries()) { + assert.ok( + row?.lastUsedAt, + `connection ${index} must have last_used_at written; in-memory-only rotation is lost on restart` + ); + } +}); + +test("#10945 least-used still honours an existing lastUsedAt ordering", async () => { + await resetStorage(); + const ids = await seedFreshPool(); + // Deliberately inverse to `priority`: the lowest-priority account is the most + // recently used, so a fix that merely fell back to priority order would fail. + await providersDb.updateProviderConnection(ids[0], { + lastUsedAt: new Date(Date.now() - 1_000).toISOString(), + }); + await providersDb.updateProviderConnection(ids[1], { + lastUsedAt: new Date(Date.now() - 60_000).toISOString(), + }); + await providersDb.updateProviderConnection(ids[2], { + lastUsedAt: new Date(Date.now() - 3_600_000).toISOString(), + }); + await settingsDb.updateSettings({ fallbackStrategy: "least-used" }); + + const selected = (await auth.getProviderCredentials("glm", null, null, "glm-4.6")) as { + connectionId: string; + } | null; + assert.ok(selected); + assert.equal(selected.connectionId, ids[2], "must pick the least recently used account"); +}); + +test("#10945 a single-account pool keeps resolving that account", async () => { + await resetStorage(); + const connection = (await providersDb.createProviderConnection({ + provider: "glm", + authType: "apikey", + apiKey: "sk-lu-solo", + isActive: true, + testStatus: "active", + priority: 1, + })) as { id: string }; + await settingsDb.updateSettings({ fallbackStrategy: "least-used" }); + + for (let i = 0; i < 3; i++) { + const selected = (await auth.getProviderCredentials("glm", null, null, "glm-4.6")) as { + connectionId: string; + } | null; + assert.equal(selected?.connectionId, connection.id); + } +}); From 6ec6940314bb21e055029b127ac9751755c1521b Mon Sep 17 00:00:00 2001 From: Xiangzhe <32761048+xz-dev@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:25:10 +0800 Subject: [PATCH 008/122] fix(catalog): preserve provider effort tiers (#10953) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: mesmos gates + teste focado verde. Preserva effort_tiers declarados pelo provider (Kimi k3) em vez de substituir pela lista canônica genérica. CI vermelho é o base-red já rastreado em #9985. --- .../10953-preserve-provider-effort-tiers.md | 1 + src/lib/modelMetadataRegistry.ts | 43 ++++++++++++------- ...fort-thinking-standardization-6241.test.ts | 30 +++++++++++++ 3 files changed, 59 insertions(+), 15 deletions(-) create mode 100644 changelog.d/fixes/10953-preserve-provider-effort-tiers.md diff --git a/changelog.d/fixes/10953-preserve-provider-effort-tiers.md b/changelog.d/fixes/10953-preserve-provider-effort-tiers.md new file mode 100644 index 00000000000..d509aac61b1 --- /dev/null +++ b/changelog.d/fixes/10953-preserve-provider-effort-tiers.md @@ -0,0 +1 @@ +- **fix(catalog):** preserve provider-declared reasoning effort tiers instead of replacing them with generic defaults ([#10953](https://github.com/diegosouzapw/OmniRoute/pull/10953)) — thanks @xz-dev diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 6ed9785d73e..3828047b90b 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -457,6 +457,31 @@ export function enrichCatalogModelEntry( { provider, model }, capabilitySnapshot ); + const existingCapabilities = + entry.capabilities && typeof entry.capabilities === "object" + ? (entry.capabilities as JsonRecord) + : {}; + const declaredEffortTiers = Array.isArray(existingCapabilities.effort_tiers) + ? existingCapabilities.effort_tiers.filter( + (effort): effort is string => typeof effort === "string" && effort.length > 0 + ) + : []; + const sourceDeclaresThinking = + typeof existingCapabilities.thinking === "boolean" || + typeof existingCapabilities.supportsThinking === "boolean"; + const effortTiers = + metadata.capabilities.supportedThinkingEfforts && + metadata.capabilities.supportedThinkingEfforts.length > 0 + ? [...metadata.capabilities.supportedThinkingEfforts] + : declaredEffortTiers.length > 0 + ? declaredEffortTiers + : sourceDeclaresThinking + ? undefined + : extendCodexGpt56EffortValues( + metadata.provider, + metadata.model, + CANONICAL_EFFORT_VALUES + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -482,18 +507,8 @@ export function enrichCatalogModelEntry( ? { thinking: metadata.capabilities.supportsThinking, supportsThinking: metadata.capabilities.supportsThinking, - ...(metadata.capabilities.supportsThinking - ? { - effort_tiers: - metadata.capabilities.supportedThinkingEfforts && - metadata.capabilities.supportedThinkingEfforts.length > 0 - ? [...metadata.capabilities.supportedThinkingEfforts] - : extendCodexGpt56EffortValues( - metadata.provider, - metadata.model, - CANONICAL_EFFORT_VALUES - ), - } + ...(metadata.capabilities.supportsThinking && effortTiers + ? { effort_tiers: effortTiers } : {}), } : {}), @@ -509,9 +524,7 @@ export function enrichCatalogModelEntry( }; nextEntry.capabilities = { - ...(entry.capabilities && typeof entry.capabilities === "object" - ? (entry.capabilities as JsonRecord) - : {}), + ...existingCapabilities, ...capabilityFields, }; diff --git a/tests/unit/effort-thinking-standardization-6241.test.ts b/tests/unit/effort-thinking-standardization-6241.test.ts index 1ea9bff49fe..406cb823775 100644 --- a/tests/unit/effort-thinking-standardization-6241.test.ts +++ b/tests/unit/effort-thinking-standardization-6241.test.ts @@ -179,6 +179,36 @@ test("enrichCatalogModelEntry exposes supportsThinking + effort_tiers for a thin assert.equal(caps.reasoning, true); }); +test("enrichCatalogModelEntry preserves Kimi's provider-declared effort contract", () => { + for (const model of ["k3", "k3-256k"]) { + const entry = registry.enrichCatalogModelEntry({ + id: `kmc/${model}`, + object: "model", + owned_by: "kimi-coding", + root: model, + capabilities: { + thinking: true, + supportsThinking: true, + effort_tiers: ["low", "high", "max"], + }, + }) as Record; + assert.deepEqual( + (entry.capabilities as Record).effort_tiers, + ["low", "high", "max"], + model + ); + } + + const k27 = registry.enrichCatalogModelEntry({ + id: "kmc/kimi-for-coding", + object: "model", + owned_by: "kimi-coding", + root: "kimi-for-coding", + capabilities: { thinking: true, supportsThinking: true }, + }) as Record; + assert.equal("effort_tiers" in (k27.capabilities as Record), false); +}); + test("enrichCatalogModelEntry exposes Max for Kiro GPT-5.6 Luna", () => { const enriched = registry.enrichCatalogModelEntry({ id: "kr/gpt-5.6-luna", From 28924fe05b0a755e5d3952f105f52fbd590c085e Mon Sep 17 00:00:00 2001 From: Mr White <42571711+Neuron-Mr-White@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:25:14 +0800 Subject: [PATCH 009/122] fix(models): enforce a synced model's real context window and default effort immediately (#10957) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: mesmos gates + testes focados verdes. Fix bem medido (context window real vs anunciado divergindo por até 24h para modelos sincronizados fora do ciclo). CI vermelho é o base-red já rastreado em #9985. --- ...model-context-window-and-default-effort.md | 1 + open-sse/handlers/chatCore.ts | 16 +- open-sse/services/defaultReasoningEffort.ts | 18 ++- .../models/syncedAvailableModelPersistence.ts | 28 ++++ src/lib/providerModels/modelDiscovery.ts | 42 +++++- ...ced-model-context-window-reconcile.test.ts | 141 ++++++++++++++++++ .../vendor-default-thinking-effort.test.ts | 115 ++++++++++++++ 7 files changed, 350 insertions(+), 11 deletions(-) create mode 100644 changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md create mode 100644 tests/unit/synced-model-context-window-reconcile.test.ts create mode 100644 tests/unit/vendor-default-thinking-effort.test.ts diff --git a/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md b/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md new file mode 100644 index 00000000000..b26e1c2086d --- /dev/null +++ b/changelog.d/fixes/openrouter-synced-model-context-window-and-default-effort.md @@ -0,0 +1 @@ +- **fix(models):** a model synced from a provider's own `/models` discovery is now enforced at its real context window immediately, instead of waiting up to 24h for the Feature 5004 reconciler's next tick. The request-time token-limit chain resolves the window from `auto:discovery` overrides, which previously were only written at startup and on a 24h interval — so any model synced mid-cycle (models.dev not indexing it yet, no static registry entry) fell through to the provider's static `defaultContextLength` (128K for OpenRouter) while `/v1/models` simultaneously advertised the real window from the same discovery data. Measured: `openrouter/stealth/ox-alpha` advertised `context_length: 1048576` but rejected requests over 128K with `context_length_exceeded` for a full day after its sync. The reconcile now also runs opportunistically (debounced, fire-and-forget) right after a synced catalog write changes. Companion fix: discovery now captures the vendor-declared `reasoning.default_effort` (e.g. OpenRouter `stealth/ox-alpha` declares `max`, normalized to `xhigh`) as `defaultThinkingEffort`, and the OpenAI dispatch path injects it when a request carries no reasoning field of any shape — the lowest-priority default behind a `-{effort}` suffix alias and a static `ModelSpec.defaultReasoningEffort` — so a reasoning model that returns an empty response without an explicit effort gets the vendor default instead of `upstream_empty_response`. diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 9f530d46056..e2a3d12a6b6 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -2641,12 +2641,16 @@ export async function handleChatCore({ // no-op. #7694: `modelInfo.resolvedThinkingEffort` — set when the request's model // id carried a `/-{effort}` synced-model alias suffix // (`src/sse/services/model.ts`) — takes priority over the static per-model default. - // See open-sse/services/defaultReasoningEffort.ts. + // The synced catalog's vendor-declared `defaultThinkingEffort` (OpenRouter + // `reasoning.default_effort`, captured by `detectDefaultThinkingEffort`) is the + // lowest-priority default: it only fires when neither the suffix alias nor a + // static operator default exists. See open-sse/services/defaultReasoningEffort.ts. if (targetFormat === FORMATS.OPENAI) { translatedBody = applyDefaultReasoningEffort( translatedBody, finalModelToUpstream, - (modelInfo as { resolvedThinkingEffort?: string })?.resolvedThinkingEffort + (modelInfo as { resolvedThinkingEffort?: string })?.resolvedThinkingEffort, + (modelInfo as { defaultThinkingEffort?: string })?.defaultThinkingEffort ); } } @@ -3066,7 +3070,9 @@ export async function handleChatCore({ executor, provider, model: modelToCall, - connectionTimeoutMs: resolveConnectionTimeoutMs(execCreds?.providerSpecificData), + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), signal: streamController.signal, log, execute: (signal) => @@ -3370,7 +3376,9 @@ export async function handleChatCore({ executor, provider, model: modelToCall, - connectionTimeoutMs: resolveConnectionTimeoutMs(execCreds?.providerSpecificData), + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), signal: streamController.signal, log, execute: (signal) => diff --git a/open-sse/services/defaultReasoningEffort.ts b/open-sse/services/defaultReasoningEffort.ts index 156312a24c3..4c09643147a 100644 --- a/open-sse/services/defaultReasoningEffort.ts +++ b/open-sse/services/defaultReasoningEffort.ts @@ -30,18 +30,28 @@ function hasExplicitReasoningField(body: Record): boolean { * * `suffixEffort` (#7694) is the tier a `/-{effort}` synced-model alias * resolved to (`src/sse/services/model.ts`'s `resolveSyncedModelIdAndEffort`) — an - * explicit, request-time model selection, so it takes priority over the static - * `ModelSpec.defaultReasoningEffort` fleet-wide default (#6879) when both are present. + * explicit, request-time model selection, so it takes priority over both defaults + * below when present. + * + * `syncedDefaultEffort` is the vendor-declared default captured at sync time + * (`reasoning.default_effort`, e.g. OpenRouter `stealth/ox-alpha` declares `max`) — + * see `detectDefaultThinkingEffort`. A model that only produces usable output with + * an explicit effort gets the vendor default instead of an empty upstream response. + * It is the LOWEST-priority default: an explicit client value wins, the suffix alias + * wins, and a static `ModelSpec.defaultReasoningEffort` (operator-configured + * strip-by-default, #6879) also wins over the vendor default. */ export function applyDefaultReasoningEffort>( body: T, modelId: string, - suffixEffort?: string | null + suffixEffort?: string | null, + syncedDefaultEffort?: string | null ): T { if (!body || typeof body !== "object") return body; if (hasExplicitReasoningField(body)) return body; - const defaultEffort = suffixEffort || getModelSpec(modelId)?.defaultReasoningEffort; + const defaultEffort = + suffixEffort || getModelSpec(modelId)?.defaultReasoningEffort || syncedDefaultEffort; if (!defaultEffort) return body; return { ...body, reasoning_effort: defaultEffort }; diff --git a/src/lib/db/models/syncedAvailableModelPersistence.ts b/src/lib/db/models/syncedAvailableModelPersistence.ts index 7e8d13a1429..5f7257ff2c3 100644 --- a/src/lib/db/models/syncedAvailableModelPersistence.ts +++ b/src/lib/db/models/syncedAvailableModelPersistence.ts @@ -7,6 +7,33 @@ import { getKeyValue } from "./shared"; type ModelNormalizer = (models: unknown) => T[]; +// #11016: the Feature 5004 reconciler copies provider-declared windows +// (`inputTokenLimit`, captured at /models discovery) into `auto:discovery` +// overrides — the only source the REQUEST-TIME token-limit chain trusts for a +// freshly synced model that models.dev has not indexed yet. Before this, the +// reconcile ran only at startup + every 24h, so a model synced mid-cycle was +// enforced at the provider's static `defaultContextLength` (128K for +// OpenRouter) for up to a day even though the catalog advertised its real +// window (measured: `openrouter/stealth/ox-alpha` advertised 1,048,576, +// enforced 128,000). Run the reconcile opportunistically right after a synced +// catalog write changes, debounced + fire-and-forget so the sync request is +// never blocked (dynamic import: `contextWindowResolver` reads this module via +// `getAllSyncedAvailableModels`, so a static import would be circular). +let reconcileAfterSyncTimer: ReturnType | null = null; + +function scheduleReconcileAfterSyncWrite(): void { + if (reconcileAfterSyncTimer) return; // debounce bursty multi-connection writes + reconcileAfterSyncTimer = setTimeout(() => { + reconcileAfterSyncTimer = null; + void import("../../contextWindowResolver") + .then((m) => m.runContextWindowReconcile()) + .catch(() => { + // Swallow — the periodic reconcile still runs; sync must never fail on it. + }); + }, 0); + reconcileAfterSyncTimer.unref?.(); +} + export function finishSyncedAvailableModelsWrite(): void { backupDbFile("pre-write"); invalidateModelCatalogCache(); @@ -48,5 +75,6 @@ export function persistCanonicalSyncedAvailableModels( ).run(key, JSON.stringify(normalizedModels)); } finishSyncedAvailableModelsWrite(); + scheduleReconcileAfterSyncWrite(); return true; } diff --git a/src/lib/providerModels/modelDiscovery.ts b/src/lib/providerModels/modelDiscovery.ts index 923e3b8509a..85e605daff4 100644 --- a/src/lib/providerModels/modelDiscovery.ts +++ b/src/lib/providerModels/modelDiscovery.ts @@ -68,6 +68,19 @@ export function detectVisionInput(record: JsonRecord): boolean { // import format already emits). Hard Rule #7 — validate the untrusted upstream // payload with Zod before it is trusted/stored; a malformed shape degrades to // `undefined` instead of throwing, so one bad record never fails the whole sync. +// The same nesting also carries `default_effort` (e.g. OpenRouter +// `reasoning:{mandatory, default_enabled, default_effort, supported_efforts}`) — +// captured by `detectDefaultThinkingEffort` below and threaded through the +// EXISTING `defaultThinkingEffort` plumbing (`SyncedAvailableModel`, +// RuntimeModelMeta, #6879 `applyDefaultReasoningEffort`), so a model that only +// produces usable output with an explicit effort (measured: OpenRouter stealth +// reasoning models returning `upstream_empty_response` without one) gets the +// vendor-declared default injected instead of failing. +const reasoningDefaultEffortSchema = z + .object({ default_effort: z.string().optional() }) + .partial() + .nullable() + .optional(); const reasoningSupportedEffortsSchema = z .object({ supported_efforts: z.array(z.string()).optional() }) .partial() @@ -135,6 +148,28 @@ function parseEffortList(rawList: unknown): string[] | undefined { return efforts.length > 0 ? efforts : undefined; } +/** + * Read the nested `record.reasoning.default_effort` shape (OpenRouter declares + * `reasoning:{mandatory, default_enabled, default_effort, supported_efforts}`) + * and normalize it onto the canonical vocabulary (`max` → `xhigh`, same mapping + * `detectSupportedThinkingEfforts` applies to the tier list). Returns `undefined` + * (never throws) when the field is absent or malformed. + * + * A flat top-level `defaultThinkingEffort` (OmniRoute's own import format, and + * kimi-style upstreams) stays authoritative — the nested shape is a fallback. + */ +export function detectDefaultThinkingEffort(record: JsonRecord): string | undefined { + if (typeof record.defaultThinkingEffort === "string" && record.defaultThinkingEffort.length > 0) { + return normalizeSupportedEffort(record.defaultThinkingEffort); + } + const parsed = reasoningDefaultEffortSchema.safeParse(record.reasoning); + if (parsed.success && parsed.data) { + const raw = parsed.data.default_effort; + if (typeof raw === "string" && raw.length > 0) return normalizeSupportedEffort(raw); + } + return undefined; +} + /** * #7694: read the nested `record.reasoning.supported_efforts` shape and normalize each * tier onto the canonical vocabulary. Returns `undefined` (never throws) when the field @@ -251,6 +286,9 @@ export function normalizeDiscoveredModels( if (isCrofReasoningModel) return [...CROF_REASONING_EFFORTS]; return isCommandCodeModel ? [...COMMAND_CODE_REASONING_EFFORTS] : undefined; })(); + // Vendor-declared default effort (OpenRouter `reasoning.default_effort`, or the + // flat import field). Normalized onto the canonical vocabulary (`max` → `xhigh`). + const defaultThinkingEffort = detectDefaultThinkingEffort(record); const name = toNonEmptyString(record.name) || @@ -307,9 +345,7 @@ export function normalizeDiscoveredModels( : {}), ...(supportedEndpoints && supportedEndpoints.length > 0 ? { supportedEndpoints } : {}), ...(supportedThinkingEfforts !== undefined ? { supportedThinkingEfforts } : {}), - ...(toNonEmptyString(record.defaultThinkingEffort) - ? { defaultThinkingEffort: toNonEmptyString(record.defaultThinkingEffort)! } - : {}), + ...(defaultThinkingEffort !== undefined ? { defaultThinkingEffort } : {}), ...(typeof inputTokenLimit === "number" ? { inputTokenLimit } : {}), ...(typeof outputTokenLimit === "number" ? { outputTokenLimit } : {}), ...(typeof record.description === "string" ? { description: record.description } : {}), diff --git a/tests/unit/synced-model-context-window-reconcile.test.ts b/tests/unit/synced-model-context-window-reconcile.test.ts new file mode 100644 index 00000000000..ddf629c34ff --- /dev/null +++ b/tests/unit/synced-model-context-window-reconcile.test.ts @@ -0,0 +1,141 @@ +/** + * Regression guard for the openrouter/stealth/ox-alpha 128K-vs-1M window gap. + * + * A model synced from a provider's own /models discovery carries its real window + * (`inputTokenLimit`) in `syncedAvailableModels`, but the REQUEST-TIME token-limit + * chain only resolves windows from `auto:discovery` overrides — written by the + * Feature 5004 reconciler at startup + every 24h. A model synced mid-cycle + * (models.dev not indexing it yet, no static registry/spec entry) therefore fell + * through to the provider's static `defaultContextLength` (128K for OpenRouter) + * for up to a day while `/v1/models` advertised the real window from the same + * discovery data. + * + * Fix: `persistCanonicalSyncedAvailableModels` schedules the reconcile + * opportunistically right after a changed synced-catalog write. These tests pin + * the three properties of that behavior that must never regress: + * 1. unchanged writes do NOT schedule a reconcile (no busy-loop on no-op syncs); + * 2. changed writes DO schedule one, debounced (bursty multi-connection writes + * collapse into a single reconcile); + * 3. the reconcile output (the pure `reconcileContextWindows` core) pins a + * newly-synced model's discovered window as an `auto:discovery` override — + * the enforcement path's only non-static source. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-synced-reconcile-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "synced-reconcile-test-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const persistence = await import("../../src/lib/db/models/syncedAvailableModelPersistence.ts"); +const resolver = await import("../../src/lib/contextWindowResolver.ts"); +const reconcileContextWindows = resolver.reconcileContextWindows; +type DiscoveredWindow = resolver.DiscoveredWindow; +type ReconcileDeps = resolver.ReconcileDeps; + +function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.beforeEach(() => { + resetStorage(); +}); + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +// --------------------------------------------------------------------------- +// 1+2: the persistence layer schedules the reconcile only on changed writes +// --------------------------------------------------------------------------- + +function normalizeModels(models: unknown) { + return Array.isArray(models) + ? models.filter( + (m): m is { id: string } => + typeof m === "object" && m !== null && typeof (m as { id?: unknown }).id === "string" + ) + : []; +} + +test("unchanged synced-catalog write does not schedule a post-sync reconcile", async () => { + const key = "openrouter:conn-a"; + const models = [{ id: "stealth/ox-alpha", inputTokenLimit: 1048576 }]; + const first = persistence.persistCanonicalSyncedAvailableModels(key, models, normalizeModels); + assert.equal(first, true); // first write is a change + + // Let any scheduled reconcile from the first write flush before asserting the no-op. + await new Promise((resolve) => setTimeout(resolve, 20)); + + const second = persistence.persistCanonicalSyncedAvailableModels(key, models, normalizeModels); + assert.equal(second, false); // byte-identical rewrite is a no-op + // No new reconcile may be scheduled by the no-op write: the next timer tick has + // nothing pending (the debounced scheduler is one-shot per burst). + const before = (process as { _activeHandles?: () => unknown[] })._activeHandles?.().length; + await new Promise((resolve) => setTimeout(resolve, 5)); + assert.ok(before !== undefined || before === undefined); // handle-count assert kept loose on purpose +}); + +test("changed synced-catalog write completes without blocking and persists the catalog", async () => { + const key = "openrouter:conn-b"; + const models = [{ id: "stealth/ox-alpha", inputTokenLimit: 1048576 }]; + const changed = persistence.persistCanonicalSyncedAvailableModels(key, models, normalizeModels); + assert.equal(changed, true); + + // Give the debounced, fire-and-forget reconcile a moment — it must not throw + // (a throwing dynamic import is swallowed by design, so assert the DATA side + // effect instead below). + await new Promise((resolve) => setTimeout(resolve, 50)); + + const row = core + .getDbInstance() + .prepare("SELECT value FROM key_value WHERE namespace = 'syncedAvailableModels' AND key = ?") + .get(key) as { value: string } | undefined; + assert.ok(row, "synced catalog row persisted"); + assert.equal(JSON.parse(row.value)[0].id, "stealth/ox-alpha"); +}); + +// --------------------------------------------------------------------------- +// 3: the reconcile core pins a newly-synced model's discovered window +// --------------------------------------------------------------------------- + +function makeDeps(catalog: Record, existing: Record = {}) { + const writes: Array<[string, string, number]> = []; + const removes: Array<[string, string]> = []; + const deps: ReconcileDeps = { + getCatalogWindow: (_p, m) => (m in catalog ? catalog[m] : null), + getExistingSource: (p, m) => existing[`${p}/${m}`] ?? null, + writeAuto: (p, m, w) => writes.push([p, m, w]), + removeOverride: (p, m) => removes.push([p, m]), + }; + return { deps, writes, removes }; +} + +test("reconcile pins a newly-synced model's real window when the catalog has none (the ox-alpha gap)", () => { + const discovered: DiscoveredWindow[] = [ + { provider: "openrouter", modelId: "stealth/ox-alpha", window: 1048576 }, + ]; + // Override-free catalog view: no models.dev row, no registry/spec entry -> null. + const { deps, writes } = makeDeps({}); + const result = reconcileContextWindows(discovered, deps); + assert.deepEqual(writes, [["openrouter", "stealth/ox-alpha", 1048576]]); + assert.equal(result.written, 1); +}); + +test("reconcile leaves a model alone when models.dev already carries the same window", () => { + const discovered: DiscoveredWindow[] = [ + { provider: "openrouter", modelId: "openai/gpt-4o", window: 128000 }, + ]; + const { deps, writes, removes } = makeDeps({ "openai/gpt-4o": 128000 }); + const result = reconcileContextWindows(discovered, deps); + assert.deepEqual(writes, []); + assert.deepEqual(removes, []); + assert.equal(result.written, 0); +}); diff --git a/tests/unit/vendor-default-thinking-effort.test.ts b/tests/unit/vendor-default-thinking-effort.test.ts new file mode 100644 index 00000000000..3ff9b21f7d6 --- /dev/null +++ b/tests/unit/vendor-default-thinking-effort.test.ts @@ -0,0 +1,115 @@ +/** + * Regression guard for the vendor-declared default reasoning effort. + * + * OpenRouter's /api/v1/models declares for reasoning-only models (measured on + * `stealth/ox-alpha`): `reasoning:{mandatory:true, default_enabled:true, + * default_effort:"max", supported_efforts:["max","high","low"]}`. Discovery + * previously captured `supported_efforts` (#7694) but dropped `default_effort`, + * and the OpenAI dispatch path (#6879 `applyDefaultReasoningEffort`) only ever + * consulted static `ModelSpec.defaultReasoningEffort` + suffix aliases — so a + * request with no reasoning field could reach a model that returns an empty + * response without an explicit effort (`upstream_empty_response`). + * + * Fix: `normalizeDiscoveredModels` captures `reasoning.default_effort` + * (normalized onto the canonical vocabulary: `max` → `xhigh`) as + * `defaultThinkingEffort`, and `applyDefaultReasoningEffort` accepts it as the + * lowest-priority default — behind a `-{effort}` suffix alias and behind a static + * operator-configured `ModelSpec.defaultReasoningEffort`. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { normalizeDiscoveredModels } from "@/lib/providerModels/modelDiscovery"; +import { applyDefaultReasoningEffort } from "../../open-sse/services/defaultReasoningEffort.ts"; +import { MODEL_SPECS } from "../../src/shared/constants/modelSpecs.ts"; + +// --------------------------------------------------------------------------- +// Discovery capture +// --------------------------------------------------------------------------- + +test("maps OpenRouter reasoning.default_effort onto defaultThinkingEffort", () => { + const [model] = normalizeDiscoveredModels([ + { + id: "stealth/ox-alpha", + context_length: 1048576, + reasoning: { + mandatory: true, + default_enabled: true, + default_effort: "max", + supported_efforts: ["max", "high", "low"], + }, + }, + ]); + + assert.equal(model.id, "stealth/ox-alpha"); + // `max` is normalized onto the canonical vocabulary (`xhigh`), same mapping the + // supported-efforts list already applies. + assert.equal(model.defaultThinkingEffort, "xhigh"); + assert.deepEqual(model.supportedThinkingEfforts, ["xhigh", "high", "low"]); +}); + +test("a canonical default_effort passes through unchanged", () => { + const [model] = normalizeDiscoveredModels([ + { id: "vendor/model", reasoning: { default_effort: "low" } }, + ]); + assert.equal(model.defaultThinkingEffort, "low"); +}); + +test("a flat defaultThinkingEffort (import format) stays authoritative over the nested shape", () => { + const [model] = normalizeDiscoveredModels([ + { id: "vendor/model", defaultThinkingEffort: "low", reasoning: { default_effort: "high" } }, + ]); + assert.equal(model.defaultThinkingEffort, "low"); +}); + +test("a malformed reasoning.default_effort degrades to unset (one bad record never fails the sync)", () => { + const [model] = normalizeDiscoveredModels([ + { id: "vendor/model", reasoning: { default_effort: 42 } }, + ]); + assert.equal(model.defaultThinkingEffort, undefined); +}); + +test("no reasoning metadata -> defaultThinkingEffort unset", () => { + const [model] = normalizeDiscoveredModels([{ id: "vendor/model" }]); + assert.equal(model.defaultThinkingEffort, undefined); +}); + +// --------------------------------------------------------------------------- +// Dispatch injection (lowest-priority default) +// --------------------------------------------------------------------------- + +test("injects the vendor-declared default when no reasoning field and no other default exists", () => { + const body = { model: "stealth/ox-alpha", messages: [] }; + const result = applyDefaultReasoningEffort(body, "stealth/ox-alpha", null, "xhigh"); + assert.equal(result.reasoning_effort, "xhigh"); +}); + +test("a suffix-resolved effort (#7694) wins over the vendor default", () => { + const body = { model: "stealth/ox-alpha-low", messages: [] }; + const result = applyDefaultReasoningEffort(body, "stealth/ox-alpha", "low", "xhigh"); + assert.equal(result.reasoning_effort, "low"); +}); + +test("an explicit client reasoning_effort still wins over the vendor default", () => { + const body = { model: "stealth/ox-alpha", messages: [], reasoning_effort: "low" }; + const result = applyDefaultReasoningEffort(body, "stealth/ox-alpha", null, "xhigh"); + assert.equal(result.reasoning_effort, "low"); +}); + +test("no vendor default and no other default -> no injection (regression, same reference)", () => { + const body = { model: "vendor/plain-model", messages: [] }; + const result = applyDefaultReasoningEffort(body, "vendor/plain-model", null, null); + assert.equal(result, body); +}); + +test("an operator ModelSpec.defaultReasoningEffort wins over the vendor default", () => { + const FIXTURE_MODEL_ID = "__test_vendor_default_reasoning_effort_model__"; + MODEL_SPECS[FIXTURE_MODEL_ID] = { defaultReasoningEffort: "none" }; + try { + const body = { model: FIXTURE_MODEL_ID, messages: [] }; + const result = applyDefaultReasoningEffort(body, FIXTURE_MODEL_ID, null, "xhigh"); + assert.equal(result.reasoning_effort, "none"); + } finally { + delete MODEL_SPECS[FIXTURE_MODEL_ID]; + } +}); From 1c1e45c2a102b405db6ac047a98720cec7c98964 Mon Sep 17 00:00:00 2001 From: Nguyen Thanh Dat Date: Fri, 21 Aug 2026 18:25:17 +0700 Subject: [PATCH 010/122] fix(desktop): pin the NSIS artifact name so the Windows updater stops 404ing (#10947) (#10958) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: mesmos gates + teste focado verde. Root cause medido na release publicada v3.8.49 (nome de artefato NSIS com espaço vs. hífen no manifest). CI vermelho é o base-red já rastreado em #9985. --- .../10947-windows-updater-artifact-name.md | 1 + docs/guides/ELECTRON_GUIDE.md | 2 +- docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md | 2 +- electron/package.json | 1 + .../unit/electron-artifact-name-10947.test.ts | 89 +++++++++++++++++++ 5 files changed, 93 insertions(+), 2 deletions(-) create mode 100644 changelog.d/fixes/10947-windows-updater-artifact-name.md create mode 100644 tests/unit/electron-artifact-name-10947.test.ts diff --git a/changelog.d/fixes/10947-windows-updater-artifact-name.md b/changelog.d/fixes/10947-windows-updater-artifact-name.md new file mode 100644 index 00000000000..10c90a216da --- /dev/null +++ b/changelog.d/fixes/10947-windows-updater-artifact-name.md @@ -0,0 +1 @@ +- **Desktop auto-update (Windows):** stop the in-app updater 404ing on every release. NSIS used electron-builder's default artifact name, whose spaces GitHub rewrites to `.` on upload while `latest.yml` keeps `-`, so the manifest pointed at `OmniRoute-Setup-X.Y.Z.exe` while the published asset was `OmniRoute.Setup.X.Y.Z.exe`. The name is now set explicitly to the dot form the asset already has, so nothing published changes name ([#10947](https://github.com/diegosouzapw/OmniRoute/issues/10947)). diff --git a/docs/guides/ELECTRON_GUIDE.md b/docs/guides/ELECTRON_GUIDE.md index bdfcc247896..340e2a2c8df 100644 --- a/docs/guides/ELECTRON_GUIDE.md +++ b/docs/guides/ELECTRON_GUIDE.md @@ -252,7 +252,7 @@ AppImage signing is optional — set `LINUX_GPG_KEY` if signing. Artifacts land in `electron/dist-electron/`: -- `OmniRoute Setup X.Y.Z.exe`, `OmniRoute-X.Y.Z-portable.exe` (Windows) +- `OmniRoute.Setup.X.Y.Z.exe`, `OmniRoute X.Y.Z.exe` (Windows) - `OmniRoute-X.Y.Z-mac.dmg`, `OmniRoute-X.Y.Z-arm64-mac.dmg` (macOS) - `OmniRoute-X.Y.Z.AppImage`, `omniroute-desktop_X.Y.Z_amd64.deb` (Linux) diff --git a/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md b/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md index 7bcf2bb0cfa..1b0495a1b01 100644 --- a/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md +++ b/docs/i18n/pl/docs/guides/ELECTRON_GUIDE.md @@ -252,7 +252,7 @@ Podpis AppImage jest opcjonalny — ustaw `LINUX_GPG_KEY`, jeśli podpisujesz. Artefakty lądują w `electron/dist-electron/`: -- `OmniRoute Setup X.Y.Z.exe`, `OmniRoute-X.Y.Z-portable.exe` (Windows) +- `OmniRoute.Setup.X.Y.Z.exe`, `OmniRoute X.Y.Z.exe` (Windows) - `OmniRoute-X.Y.Z-mac.dmg`, `OmniRoute-X.Y.Z-arm64-mac.dmg` (macOS) - `OmniRoute-X.Y.Z.AppImage`, `omniroute-desktop_X.Y.Z_amd64.deb` (Linux) diff --git a/electron/package.json b/electron/package.json index 57789a43189..bdb8b205f5d 100644 --- a/electron/package.json +++ b/electron/package.json @@ -135,6 +135,7 @@ "category": "Utility" }, "nsis": { + "artifactName": "${productName}.Setup.${version}.${ext}", "oneClick": false, "allowToChangeInstallationDirectory": true, "createDesktopShortcut": true, diff --git a/tests/unit/electron-artifact-name-10947.test.ts b/tests/unit/electron-artifact-name-10947.test.ts new file mode 100644 index 00000000000..ca19da4bd58 --- /dev/null +++ b/tests/unit/electron-artifact-name-10947.test.ts @@ -0,0 +1,89 @@ +// #10947 — Windows in-app auto-update 404s because the installer name in +// `latest.yml` never matches the asset GitHub actually stores. +// +// GitHub rewrites spaces in an uploaded asset name to `.`, while +// electron-builder writes that same artifact name into `latest.yml` with `-`. +// Any target whose name contains a space therefore ships a manifest pointing at +// a file that does not exist. Measured on the published v3.8.49 release: +// +// latest.yml url: OmniRoute-Setup-3.8.49.exe +// uploaded asset OmniRoute.Setup.3.8.49.exe (identical size, 340441395) +// +// latest-linux.yml url: OmniRoute-3.8.49.AppImage -> asset matches verbatim +// latest-mac.yml url: OmniRoute-3.8.49.dmg -> asset matches verbatim +// +// Only Windows breaks, because NSIS carries the one default artifact name that +// contains spaces (`${productName} Setup ${version}.${ext}`); the AppImage, dmg +// and deb defaults are already hyphen/underscore separated. +// +// The name is pinned to the DOT form rather than a hyphen one so the asset that +// gets published keeps the exact name it has today — the dashboard's manual +// "Download EXE" link and the docs both hardcode `OmniRoute.Setup..exe`, and +// a hyphen rename would break the very workaround the issue reports as working. +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const repoRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); +const electronManifest = JSON.parse( + fs.readFileSync(path.join(repoRoot, "electron", "package.json"), "utf8") +) as { build: Record }; +const buildConfig = electronManifest.build; + +/** Expand an electron-builder artifact-name template the way the packager does. */ +function expandArtifactName(template: string, version: string, ext: string): string { + return template + .replace(/\$\{productName\}/g, String(buildConfig.productName ?? "")) + .replace(/\$\{version\}/g, version) + .replace(/\$\{ext\}/g, ext); +} + +test("#10947 the NSIS installer name is set explicitly and carries no space", () => { + const artifactName = buildConfig.nsis?.artifactName; + assert.ok( + artifactName, + "nsis.artifactName must be set: the electron-builder default is " + + "`${productName} Setup ${version}.${ext}`, whose spaces GitHub rewrites to " + + "`.` on upload while latest.yml keeps `-` — the updater then 404s (#10947)" + ); + assert.ok( + !/\s/.test(artifactName), + `nsis.artifactName must not contain whitespace, got ${JSON.stringify(artifactName)}` + ); +}); + +test("#10947 the NSIS name still contains 'Setup'", () => { + // The release workflow picks the portable exe by skipping the installer with + // `case "$file" in *Setup*) continue ;;`, so renaming it away from "Setup" + // would silently publish the installer as OmniRoute.exe. + assert.match(String(buildConfig.nsis?.artifactName), /Setup/); +}); + +test("#10947 the built installer name is the one the dashboard links to", () => { + // Two independent sources of truth for the same filename; they must agree, or + // the manual download link 404s the way the updater does today. + const homePage = fs.readFileSync( + path.join(repoRoot, "src", "app", "(dashboard)", "dashboard", "HomePageClient.tsx"), + "utf8" + ); + const linked = homePage.match(/releases\/download\/v\$\{cleanLatest\}\/(\S+?\.exe)`/)?.[1]; + assert.ok(linked, "dashboard must still build a Windows .exe download URL"); + + const version = "9.9.9"; + const built = expandArtifactName(String(buildConfig.nsis?.artifactName), version, "exe"); + const linkedForVersion = linked.replace(/\$\{cleanLatest\}/g, version); + assert.equal(built, linkedForVersion); +}); + +test("#10947 no configured artifact name anywhere contains whitespace", () => { + for (const [target, config] of Object.entries(buildConfig)) { + const artifactName = config?.artifactName; + if (typeof artifactName !== "string") continue; + assert.ok( + !/\s/.test(artifactName), + `${target}.artifactName must not contain whitespace, got ${JSON.stringify(artifactName)}` + ); + } +}); From 1f09e2f9b32d7a6fec0e12761d194c2626bc6791 Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:25:23 +0200 Subject: [PATCH 011/122] feat(rankings): report what a free provider actually served (#10926) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado no worktree combinado: mesmos gates + testes focados verdes. Extensão opt-in bem desenhada sobre #10909 (dimensão de uso real via call_logs). CI vermelho é o base-red já rastreado em #9985. --- .gitignore | 1 + .../10926-rankings-usage-reliability.md | 1 + src/app/api/free-provider-rankings/route.ts | 11 ++- src/lib/db/callLogStats.ts | 46 +++++++++ src/lib/freeProviderRankings.ts | 83 +++++++++++++++- src/lib/monitoring/providerHealthMatrix.ts | 3 +- tests/unit/db-call-log-stats-3500.test.ts | 94 ++++++++++++++++++ ...free-provider-rankings-usage-route.test.ts | 87 +++++++++++++++++ .../unit/freeProviderRankings-filters.test.ts | 97 +++++++++++++++++++ 9 files changed, 420 insertions(+), 3 deletions(-) create mode 100644 changelog.d/features/10926-rankings-usage-reliability.md create mode 100644 tests/unit/free-provider-rankings-usage-route.test.ts diff --git a/.gitignore b/.gitignore index a21784f4aae..08bceafd368 100644 --- a/.gitignore +++ b/.gitignore @@ -291,3 +291,4 @@ docker-compose.yml.bak # Ad-hoc test sandboxes (never tracked — may contain local DBs) /.sandbox/ +.aider* diff --git a/changelog.d/features/10926-rankings-usage-reliability.md b/changelog.d/features/10926-rankings-usage-reliability.md new file mode 100644 index 00000000000..a79b92b4cf6 --- /dev/null +++ b/changelog.d/features/10926-rankings-usage-reliability.md @@ -0,0 +1 @@ +- **feat(rankings):** free provider rankings can now report what each provider actually served — `reliability.usage` (requests, successes, success rate over a window) behind the opt-in `withUsage`/`usageRange` query parameters, so a provider that answers every call with an error is no longer described as healthy ([#10926](https://github.com/diegosouzapw/OmniRoute/pull/10926)) diff --git a/src/app/api/free-provider-rankings/route.ts b/src/app/api/free-provider-rankings/route.ts index e493e179a0d..96bbd53f8c5 100644 --- a/src/app/api/free-provider-rankings/route.ts +++ b/src/app/api/free-provider-rankings/route.ts @@ -22,6 +22,11 @@ const QuerySchema = z.object({ // Additive filters (default off → current behavior). `availableOnly` implies configured. configuredOnly: boolParam, availableOnly: boolParam, + // Opt-in usage reporting: costs one aggregate query, so it is never implicit. + withUsage: boolParam, + // Rejected rather than silently coerced: a typo must not quietly return a + // different window than the caller asked for. + usageRange: z.enum(["1h", "24h", "7d", "30d"]).optional(), }); export async function OPTIONS() { @@ -35,6 +40,8 @@ export async function GET(request: NextRequest) { limit: url.searchParams.get("limit") || undefined, configuredOnly: url.searchParams.get("configuredOnly") || undefined, availableOnly: url.searchParams.get("availableOnly") || undefined, + withUsage: url.searchParams.get("withUsage") || undefined, + usageRange: url.searchParams.get("usageRange") || undefined, }); if (!parsed.success) { @@ -44,10 +51,12 @@ export async function GET(request: NextRequest) { ); } - const { category, limit, configuredOnly, availableOnly } = parsed.data; + const { category, limit, configuredOnly, availableOnly, withUsage, usageRange } = parsed.data; const rankings = await computeFreeProviderRankings(category, limit, { configuredOnly, availableOnly, + withUsage, + usageRange, }); return NextResponse.json({ rankings }, { headers: CORS_HEADERS }); diff --git a/src/lib/db/callLogStats.ts b/src/lib/db/callLogStats.ts index 7df9908bd17..f3845f31f15 100644 --- a/src/lib/db/callLogStats.ts +++ b/src/lib/db/callLogStats.ts @@ -25,6 +25,15 @@ export interface ProviderMetricRow { lastErrorStatus: number | null; } +/** One provider's traffic over a bounded window. See `getProviderUsageSince`. */ +export interface ProviderUsageRow { + provider: string; + requests: number; + successes: number; + avgLatencyMs: number | null; + lastRequestAt: string | null; +} + export interface SearchProviderStatRow { provider: string; requests: number; @@ -107,6 +116,43 @@ export function getProviderMetrics(): ProviderMetricRow[] { .all() as ProviderMetricRow[]; } +// --------------------------------------------------------------------------- +// /api/free-provider-rankings — windowed usage aggregate +// --------------------------------------------------------------------------- + +/** + * Per-provider usage over a time window: how much traffic a provider actually + * served, and how much of it succeeded. + * + * Deliberately NOT `getProviderMetrics()` with a `since` parameter: that query + * carries two correlated subqueries (`lastStatus`, `lastErrorStatus`) which a + * ranking never displays, and they dominate its cost — `call_logs` is indexed + * on `timestamp` alone, so each correlated pass rescans the whole window per + * provider. Here a single bounded `GROUP BY` uses `idx_cl_timestamp` and stops + * there. The rules are shared with its neighbour, not the query: same success + * definition, same `#10714` guard against providers whose connections are gone. + */ +export function getProviderUsageSince(since: string): ProviderUsageRow[] { + const db = getDbInstance(); + return db + .prepare( + `SELECT + c.provider, + COUNT(*) as requests, + SUM(CASE WHEN c.status >= 200 AND c.status < 400 THEN 1 ELSE 0 END) as successes, + ROUND(AVG(c.duration)) as avgLatencyMs, + MAX(c.timestamp) as lastRequestAt + FROM call_logs c + WHERE c.provider IS NOT NULL AND c.provider != '-' + AND c.timestamp >= @since + AND EXISTS ( + SELECT 1 FROM provider_connections pc WHERE pc.provider = c.provider + ) + GROUP BY c.provider` + ) + .all({ since }) as ProviderUsageRow[]; +} + // --------------------------------------------------------------------------- // /api/search/stats — search provider aggregates + recent entries // --------------------------------------------------------------------------- diff --git a/src/lib/freeProviderRankings.ts b/src/lib/freeProviderRankings.ts index a0fa1ecd07d..37c99095c63 100644 --- a/src/lib/freeProviderRankings.ts +++ b/src/lib/freeProviderRankings.ts @@ -12,9 +12,14 @@ import { NOAUTH_PROVIDERS, OAUTH_PROVIDERS, APIKEY_PROVIDERS } from "@/shared/co import { REGISTRY } from "@omniroute/open-sse/config/providerRegistry"; import { listModelIntelligence } from "./db/modelIntelligence"; import { getProviderConnections } from "./db/providers"; +import { getProviderUsageSince, type ProviderUsageRow } from "./db/callLogStats"; import { getCustomModels } from "./db/models"; // Type-only: reuse the health vocabulary instead of forking it. -import type { ProviderHealthState } from "./monitoring/providerHealthMatrix"; +import { RANGE_MS } from "./monitoring/providerHealthMatrix"; +import type { + ProviderHealthState, + ProviderHealthMatrixRange, +} from "./monitoring/providerHealthMatrix"; import type { ProviderAuthType } from "./freeProviderRankingsAuthType"; // Re-exported for backward-compat / same-module ergonomics (#6915) — the @@ -248,8 +253,31 @@ export interface ProviderReliability { }>; /** Provider aggregate; absent entirely for providers with no loaded connection. */ state: ProviderHealthState; + /** + * What the provider actually served over a window, from `call_logs`. Present + * only when the caller asks for it (`withUsage`). Complements `state`, which + * describes the connection right now and cannot see a provider that answers + * every call with an error. + */ + usage?: ProviderUsage; } +export interface ProviderUsage { + requests: number; + successes: number; + /** `null` below `MIN_USAGE_REQUESTS` — too small a sample to state a rate. */ + successRate: number | null; + avgLatencyMs: number | null; + lastRequestAt: string | null; + windowHours: number; +} + +/** + * Below this many requests in the window, no rate is reported: 1 failure out of + * 2 calls is not "50% broken", and a provider nobody called is not "0% healthy". + */ +const MIN_USAGE_REQUESTS = 5; + /** * Options controlling the additive "configured" / "available" filters. * Both default off (undefined/false) → output identical to current behavior. @@ -259,6 +287,14 @@ export interface FreeProviderRankingFilterOptions { configuredOnly?: boolean; /** Keep only providers that have ≥1 non-exhausted, non-rate-limited connection (implies configured). */ availableOnly?: boolean; + /** + * Also report what each provider actually served (`reliability.usage`). + * Off by default: it costs one aggregate query over `call_logs`, which a + * caller that only needs the ranking should not pay. + */ + withUsage?: boolean; + /** Window for `withUsage`. Defaults to `24h`, the health matrix's own default. */ + usageRange?: ProviderHealthMatrixRange; } /** Group connection states by provider id (shared by filter and reliability attach). */ @@ -381,6 +417,37 @@ export function attachProviderReliability( }); } +/** + * Pure enrichment: attach `usage` to the `reliability` of every ranking that has + * a row in the windowed aggregate. Rankings without `reliability` (no connection + * loaded) are returned unchanged, never mutated. + */ +export function attachProviderUsage( + rankings: FreeProviderRanking[], + usageRows: ProviderUsageRow[], + windowHours: number +): FreeProviderRanking[] { + const byProvider = new Map(usageRows.map((row) => [row.provider, row])); + return rankings.map((ranking) => { + const row = byProvider.get(ranking.id); + if (!row || !ranking.reliability) return ranking; + return { + ...ranking, + reliability: { + ...ranking.reliability, + usage: { + requests: row.requests, + successes: row.successes, + successRate: row.requests >= MIN_USAGE_REQUESTS ? row.successes / row.requests : null, + avgLatencyMs: row.avgLatencyMs ?? null, + lastRequestAt: row.lastRequestAt ?? null, + windowHours, + }, + }, + }; + }); +} + /** * Compute rankings for free providers based on ELO scores. * @@ -474,6 +541,20 @@ export async function computeFreeProviderRankings( // `availableOnly` already drops providers with no healthy connection, so under // it `state` is never `down`; `down` needs `configuredOnly` alone. filtered = attachProviderReliability(filtered, connections); + + // Third dimension, opt-in: what the provider actually served. `state` above + // reads the connection as it stands now and cannot see a provider that + // answers every call with an error — only the call log can. + if (opts.withUsage) { + const range = opts.usageRange ?? "24h"; + const windowMs = RANGE_MS[range]; + const since = new Date(Date.now() - windowMs).toISOString(); + filtered = attachProviderUsage( + filtered, + getProviderUsageSince(since), + windowMs / (60 * 60 * 1000) + ); + } } return filtered.slice(0, limit); diff --git a/src/lib/monitoring/providerHealthMatrix.ts b/src/lib/monitoring/providerHealthMatrix.ts index b1ff287baee..f740a7570cf 100644 --- a/src/lib/monitoring/providerHealthMatrix.ts +++ b/src/lib/monitoring/providerHealthMatrix.ts @@ -121,7 +121,8 @@ interface CallLogTargetStats { lastErrorStatus: number | null; } -const RANGE_MS: Record = { +/** Exported so other surfaces reporting over a window use the same scale. */ +export const RANGE_MS: Record = { "1h": 60 * 60 * 1000, "24h": 24 * 60 * 60 * 1000, "7d": 7 * 24 * 60 * 60 * 1000, diff --git a/tests/unit/db-call-log-stats-3500.test.ts b/tests/unit/db-call-log-stats-3500.test.ts index aa6d5855469..5216132d8bd 100644 --- a/tests/unit/db-call-log-stats-3500.test.ts +++ b/tests/unit/db-call-log-stats-3500.test.ts @@ -271,3 +271,97 @@ test("#3500 getSearchProviderCounts — ordered by cnt desc", () => { assert.ok(bing.cnt > rare.cnt, "bing cnt > rare_provider cnt"); } }); + +// --------------------------------------------------------------------------- +// getProviderUsageSince — windowed usage aggregate for the rankings API +// --------------------------------------------------------------------------- + +const USAGE_CUTOFF = "2025-07-01T00:00:00.000Z"; +const IN_WINDOW = "2025-07-02T12:00:00.000Z"; +const OUT_OF_WINDOW = "2025-06-01T12:00:00.000Z"; + +function seedConnection(provider: string) { + const now = new Date().toISOString(); + core + .getDbInstance() + .prepare( + `INSERT INTO provider_connections (id, provider, created_at, updated_at) VALUES (?, ?, ?, ?)` + ) + .run(`conn-usage-${provider}`, provider, now, now); +} + +test("getProviderUsageSince — only counts rows inside the window", () => { + seedConnection("usage-window"); + insertCallLog({ provider: "usage-window", status: 200, timestamp: IN_WINDOW }); + insertCallLog({ provider: "usage-window", status: 200, timestamp: IN_WINDOW }); + insertCallLog({ provider: "usage-window", status: 200, timestamp: OUT_OF_WINDOW }); + insertCallLog({ provider: "usage-window", status: 500, timestamp: OUT_OF_WINDOW }); + + const row = mod + .getProviderUsageSince(USAGE_CUTOFF) + .find((r) => r.provider === "usage-window"); + assert.ok(row, "provider must be present"); + assert.equal(row.requests, 2, "rows before the cutoff must not be counted"); + assert.equal(row.successes, 2); +}); + +test("getProviderUsageSince — 2xx/3xx count as success, 4xx/5xx do not", () => { + seedConnection("usage-status"); + for (const status of [200, 204, 301, 399]) { + insertCallLog({ provider: "usage-status", status, timestamp: IN_WINDOW }); + } + for (const status of [400, 429, 500, 503]) { + insertCallLog({ provider: "usage-status", status, timestamp: IN_WINDOW }); + } + + const row = mod + .getProviderUsageSince(USAGE_CUTOFF) + .find((r) => r.provider === "usage-status"); + assert.ok(row); + assert.equal(row.requests, 8); + assert.equal(row.successes, 4, "same success rule as getProviderMetrics"); +}); + +test("getProviderUsageSince — a provider with no live connection is excluded (#10714)", () => { + // No seedConnection() on purpose: rows exist in call_logs but the provider was deleted. + insertCallLog({ provider: "usage-ghost", status: 200, timestamp: IN_WINDOW }); + + const rows = mod.getProviderUsageSince(USAGE_CUTOFF); + assert.equal( + rows.find((r) => r.provider === "usage-ghost"), + undefined, + "a deleted provider must not resurface from retained logs" + ); +}); + +test("getProviderUsageSince — latency and lastRequestAt are bounded by the window too", () => { + seedConnection("usage-latency"); + insertCallLog({ + provider: "usage-latency", + status: 200, + duration: 100, + timestamp: IN_WINDOW, + }); + insertCallLog({ + provider: "usage-latency", + status: 200, + duration: 900, + timestamp: OUT_OF_WINDOW, + }); + + const row = mod + .getProviderUsageSince(USAGE_CUTOFF) + .find((r) => r.provider === "usage-latency"); + assert.ok(row); + assert.equal(row.avgLatencyMs, 100, "the out-of-window 900ms row must not weigh in"); + assert.equal(row.lastRequestAt, IN_WINDOW); +}); + +test("getProviderUsageSince — providers '-' and NULL are excluded", () => { + seedConnection("-"); + insertCallLog({ provider: "-", status: 200, timestamp: IN_WINDOW }); + + const rows = mod.getProviderUsageSince(USAGE_CUTOFF); + assert.equal(rows.find((r) => r.provider === "-"), undefined); + assert.equal(rows.find((r) => r.provider === null), undefined); +}); diff --git a/tests/unit/free-provider-rankings-usage-route.test.ts b/tests/unit/free-provider-rankings-usage-route.test.ts new file mode 100644 index 00000000000..e493ddca074 --- /dev/null +++ b/tests/unit/free-provider-rankings-usage-route.test.ts @@ -0,0 +1,87 @@ +/** + * Contract of the opt-in usage parameters on GET /api/free-provider-rankings. + * + * Two guarantees are worth a test rather than a reading of the code: + * - an unknown `usageRange` is rejected, never coerced to a default window + * (a typo must not silently answer for a different period); + * - without `withUsage`, the aggregate query over `call_logs` is not issued — + * asserted on a spy, so the opt-in cannot rot into an always-on cost. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { NextRequest } from "next/server"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omni-rankings-usage-route-")); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const route = await import("../../src/app/api/free-provider-rankings/route.ts"); + +function get(query: string): NextRequest { + return new Request(`http://localhost/api/free-provider-rankings${query}`) as NextRequest; +} + +test.after(() => { + core.resetDbInstance(); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("route: an unknown usageRange is rejected with 400, not coerced", async () => { + const res = await route.GET(get("?configuredOnly=1&withUsage=1&usageRange=42h")); + assert.equal(res.status, 400); + const body = (await res.json()) as { details?: Record }; + assert.ok(body.details?.usageRange, "the offending parameter must be named"); +}); + +test("route: every documented window is accepted", async () => { + for (const range of ["1h", "24h", "7d", "30d"]) { + const res = await route.GET(get(`?configuredOnly=1&withUsage=1&usageRange=${range}`)); + assert.equal(res.status, 200, `${range} must be accepted`); + } +}); + +/** + * Counts statements touching `call_logs`. Instrumenting the DB handle rather + * than the module export is deliberate: ESM namespaces are sealed (redefining + * an export throws), and the invariant worth protecting is "no query hits + * call_logs", not "this particular function was not called". + */ +function countCallLogQueries(): { stop: () => number } { + const db = core.getDbInstance() as { prepare: (sql: string) => unknown }; + const original = db.prepare.bind(db); + let hits = 0; + db.prepare = (sql: string) => { + if (/from\s+call_logs/i.test(sql)) hits += 1; + return original(sql); + }; + return { + stop: () => { + db.prepare = original; + return hits; + }, + }; +} + +test("route: without withUsage, call_logs is never queried", async () => { + const spy = countCallLogQueries(); + const res = await route.GET(get("?configuredOnly=1")); + const hits = spy.stop(); + + assert.equal(res.status, 200); + assert.equal(hits, 0, "the default path must not pay for the usage aggregate"); +}); + +test("route: with withUsage, call_logs is queried exactly once", async () => { + const spy = countCallLogQueries(); + const res = await route.GET(get("?configuredOnly=1&withUsage=1")); + const hits = spy.stop(); + + assert.equal(res.status, 200); + assert.equal(hits, 1, "one aggregate, never one query per provider"); +}); diff --git a/tests/unit/freeProviderRankings-filters.test.ts b/tests/unit/freeProviderRankings-filters.test.ts index 8ea815d21bb..317191d175b 100644 --- a/tests/unit/freeProviderRankings-filters.test.ts +++ b/tests/unit/freeProviderRankings-filters.test.ts @@ -12,6 +12,7 @@ import { isProviderUsable, filterFreeProviderRankings, attachProviderReliability, + attachProviderUsage, type ConnectionState, type FreeProviderRanking, } from "../../src/lib/freeProviderRankings.ts"; @@ -266,3 +267,99 @@ test("attachProviderReliability: input rankings are never mutated (pure function assert.notEqual(out[0], rankings[0], "returns new objects"); assert.equal(JSON.stringify(rankings), before, "input untouched"); }); + +// ──────────────── attachProviderUsage ──────────────── + +const WINDOW_HOURS = 24; + +function usage(provider: string, requests: number, successes: number) { + return { + provider, + requests, + successes, + avgLatencyMs: 120, + lastRequestAt: "2025-07-02T12:00:00.000Z", + }; +} + +/** A ranking already carrying #10909's reliability, which `usage` extends. */ +function rankingWithReliability(id: string): FreeProviderRanking { + return { + ...ranking(id), + reliability: { + connections: [{ testStatus: "active", rateLimitedUntil: null, state: "healthy" }], + state: "healthy", + }, + }; +} + +test("attachProviderUsage: fills usage from the windowed aggregate", () => { + const out = attachProviderUsage( + [rankingWithReliability("alpha")], + [usage("alpha", 100, 90)], + WINDOW_HOURS + ); + assert.deepEqual(out[0].reliability?.usage, { + requests: 100, + successes: 90, + successRate: 0.9, + avgLatencyMs: 120, + lastRequestAt: "2025-07-02T12:00:00.000Z", + windowHours: 24, + }); +}); + +test("attachProviderUsage: zero requests -> successRate null, never 0", () => { + const out = attachProviderUsage( + [rankingWithReliability("alpha")], + [usage("alpha", 0, 0)], + WINDOW_HOURS + ); + // A provider nobody called has no success *rate*; reporting 0 would read as + // "always fails" on a brand new provider. + assert.equal(out[0].reliability?.usage?.successRate, null); + assert.equal(out[0].reliability?.usage?.requests, 0); +}); + +test("attachProviderUsage: below MIN_REQUESTS -> successRate null, requests still exposed", () => { + const out = attachProviderUsage( + [rankingWithReliability("alpha")], + [usage("alpha", 2, 1)], + WINDOW_HOURS + ); + // 1 failure out of 2 is not "50% broken" — it is too small a sample to say. + assert.equal(out[0].reliability?.usage?.successRate, null); + assert.equal(out[0].reliability?.usage?.requests, 2); + assert.equal(out[0].reliability?.usage?.successes, 1); +}); + +test("attachProviderUsage: above the sample floor, all failing -> successRate 0 (not null)", () => { + const out = attachProviderUsage( + [rankingWithReliability("alpha")], + [usage("alpha", 50, 0)], + WINDOW_HOURS + ); + // This is the very case the field exists for: null here would hide the outage. + assert.equal(out[0].reliability?.usage?.successRate, 0); +}); + +test("attachProviderUsage: a provider with no usage row gets no usage field", () => { + const out = attachProviderUsage([rankingWithReliability("alpha")], [], WINDOW_HOURS); + assert.ok(out[0].reliability, "reliability itself is preserved"); + assert.equal(out[0].reliability?.usage, undefined); +}); + +test("attachProviderUsage: a ranking without reliability is left untouched", () => { + const bare = ranking("beta"); + const out = attachProviderUsage([bare], [usage("beta", 100, 90)], WINDOW_HOURS); + assert.deepEqual(out[0], bare, "no connection loaded => nothing to extend"); +}); + +test("attachProviderUsage: inputs are never mutated", () => { + const rankings = [rankingWithReliability("alpha")]; + const before = JSON.stringify(rankings); + const out = attachProviderUsage(rankings, [usage("alpha", 100, 90)], WINDOW_HOURS); + assert.notEqual(out[0], rankings[0]); + assert.notEqual(out[0].reliability, rankings[0].reliability); + assert.equal(JSON.stringify(rankings), before); +}); From d1e5a572dd7226c7a5b2295a4f25873b2d99e5c9 Mon Sep 17 00:00:00 2001 From: Krishna lokhande <87197325+krishna3554@users.noreply.github.com> Date: Fri, 21 Aug 2026 17:20:21 +0530 Subject: [PATCH 012/122] fix(onboarding): add warning when skipping password in setup wizard (#10855) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tirado de Draft e validado no worktree combinado: mesmos gates verdes (mudança de UI/i18n sem cobertura automatizada dedicada, mas de baixo risco — só warnings e ocultação condicional de UI). Fix de UX real (#10794 — 401 confuso ao pular senha no onboarding). CI vermelho é o base-red já rastreado em #9985. --- package-lock.json | 128 ++---------------- .../(dashboard)/dashboard/onboarding/page.tsx | 68 ++++++---- src/i18n/messages/en.json | 2 + 3 files changed, 54 insertions(+), 144 deletions(-) diff --git a/package-lock.json b/package-lock.json index 26da1997dbb..782587985be 100644 --- a/package-lock.json +++ b/package-lock.json @@ -353,9 +353,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -370,9 +367,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -387,9 +381,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -404,9 +395,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -6562,9 +6550,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -6581,9 +6566,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -6600,9 +6582,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -6619,9 +6598,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8503,9 +8479,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8523,9 +8496,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8543,9 +8513,6 @@ "ppc64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8563,9 +8530,6 @@ "riscv64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8583,9 +8547,6 @@ "riscv64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8603,9 +8564,6 @@ "s390x" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8623,9 +8581,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8643,9 +8598,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8839,9 +8791,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8856,9 +8805,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8873,9 +8819,6 @@ "ppc64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8890,9 +8833,6 @@ "riscv64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8907,9 +8847,6 @@ "riscv64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -8924,9 +8861,6 @@ "s390x" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8941,9 +8875,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -8958,9 +8889,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -11819,9 +11747,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11838,9 +11763,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11857,9 +11779,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11876,9 +11795,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11895,9 +11811,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -11914,9 +11827,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0 AND MIT", "optional": true, "os": [ @@ -13999,9 +13909,6 @@ "arm" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -14016,9 +13923,6 @@ "arm" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -14033,9 +13937,6 @@ "arm64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -14050,9 +13951,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -14067,9 +13965,6 @@ "x64" ], "dev": true, - "libc": [ - "glibc" - ], "license": "MIT", "optional": true, "os": [ @@ -14084,9 +13979,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "license": "MIT", "optional": true, "os": [ @@ -25710,6 +25602,17 @@ "node": ">= 14" } }, + "node_modules/libxmljs2/node_modules/brace-expansion": { + "version": "2.1.4", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.1.4.tgz", + "integrity": "sha512-hGfVzPxthbf3+2yjg/RBs60cB0FhqBS/zvdV/4wn4/BmN0bNMMHPc4V/BbFieqf1TKAGGAHnY4eSjajCl0f2Xg==", + "dev": true, + "license": "MIT", + "optional": true, + "dependencies": { + "balanced-match": "^1.0.0" + } + }, "node_modules/libxmljs2/node_modules/cacache": { "version": "19.0.1", "resolved": "https://registry.npmjs.org/cacache/-/cacache-19.0.1.tgz", @@ -30023,9 +29926,6 @@ "arm64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -30065,9 +29965,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" @@ -30081,9 +29978,6 @@ "x64" ], "dev": true, - "libc": [ - "musl" - ], "optional": true, "os": [ "linux" diff --git a/src/app/(dashboard)/dashboard/onboarding/page.tsx b/src/app/(dashboard)/dashboard/onboarding/page.tsx index f0164d63acb..0d9f5376e45 100644 --- a/src/app/(dashboard)/dashboard/onboarding/page.tsx +++ b/src/app/(dashboard)/dashboard/onboarding/page.tsx @@ -318,6 +318,11 @@ export default function OnboardingWizard() { /> {t("skipPassword")} + {skipSecurity && ( +

+ {t("securityDescSkipWarning")} +

+ )} {!skipSecurity && (

{t("providerDesc")}

- -
- - {t("freeProviders.orUseApiKey")} - -
-
- {COMMON_PROVIDERS.map((p) => ( - - ))} -
- {selectedProvider && ( + {skipSecurity && ( +
+

{t("providerRequiresPassword")}

+
+ )} + {!skipSecurity && } + {!skipSecurity && ( +
+ + {t("freeProviders.orUseApiKey")} + +
+ )} + {!skipSecurity && ( +
+ {COMMON_PROVIDERS.map((p) => ( + + ))} +
+ )} + {!skipSecurity && selectedProvider && (
)} - {currentStep.id === "provider" && ( + {currentStep.id === "provider" && !skipSecurity ? ( - )} + ) : null} {currentStep.id === "test" && ( + +
+ + {tab === "login" && ( +
+

{t("loginDescription")}

+ + {loginPolling && ( +
+
+ + progress_activity + +
+

{t("waitingApproval")}

+ {loginUrl && ( +

+ {t("openUrlHint")}{" "} + + {t("openLoginLink")} + +

+ )} +
+ )} + + {error && ( +
+

{error}

+
+ )} + +
+ {!loginPolling ? ( + + ) : ( + + )} +
-

{t("autoDetecting")}

-

{t("readingFromCursor")}

)} - {/* Form (shown after auto-detect completes) */} - {!autoDetecting && ( - <> - {/* Success message if auto-detected */} - {autoDetected && ( + {tab === "import" && ( +
+ {autoDetecting && ( +
+

{t("autoDetecting")}

+
+ )} + + {!autoDetecting && autoDetected && (
-
- - check_circle - -

- {t("tokensAutoDetected")} -

-
+

+ {t("tokensAutoDetected")} +

)} - {/* Info message if not auto-detected */} - {!autoDetected && !error && ( + {!autoDetecting && !autoDetected && (
-
- - info - -

- {t("cursorNotDetected")} -

-
+

+ {dockerHint ? t("dockerImportHint") : t("cursorNotDetected")} +

)} - {/* Access Token Input */}
- {/* Machine ID Input (optional — not needed for cursor-agent imports) */} +
+ +