diff --git a/.env.example b/.env.example index 2f7da81f9ae..52150563540 100644 --- a/.env.example +++ b/.env.example @@ -1161,9 +1161,18 @@ PROVIDER_LIMITS_SYNC_SPACING_MS=1500 #OMNIROUTE_WAL_GUARD_MAX_MB=256 # Minimum rows a cleanup must delete before the post-cleanup VACUUM runs. Default: 1000. -# 0 always vacuums when rows were freed. Used by: src/lib/db/cleanup.ts. +# 0 always vacuums when rows were freed. The post-cleanup VACUUM also runs when the +# reclaimable-space threshold below is met, whichever comes first (either signal fires it). +# Used by: src/lib/db/cleanup.ts::shouldVacuumAfterCleanup(). #OMNIROUTE_VACUUM_MIN_DELETED_ROWS=1000 +# Minimum reclaimable space (MB) that alone justifies a full-database VACUUM after a +# cleanup, even when the row-count threshold above was not met (a handful of oversized +# blob rows can free far more space than thousands of tiny rows). VACUUM is synchronous +# and blocks the entire process. Default: 100. 0 always vacuums after any deletion. +# Used by: src/lib/db/cleanup.ts::getVacuumMinReclaimableBytes(). +#OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB=100 + # Explicit path to sql-wasm.wasm for the sql.js fallback adapter. Default: auto-detect. # Used by: src/lib/db/adapters/sqljsAdapter.ts. #OMNIROUTE_SQLJS_WASM_PATH= diff --git a/changelog.d/features/13448-adaptive-reasoning-effort.md b/changelog.d/features/13448-adaptive-reasoning-effort.md new file mode 100644 index 00000000000..04d464b6198 --- /dev/null +++ b/changelog.d/features/13448-adaptive-reasoning-effort.md @@ -0,0 +1 @@ +- **feat(reasoning):** adaptive reasoning effort (`auto`) — the gateway resolves the thinking budget per user turn from deterministic request-shape signals (stateless per-turn pin) instead of forwarding a literal `auto`, applied at the gateway pre-translation for any harness (Claude Code, Cursor, Codex, opencode, Hermes) whose request dispatches to an OpenAI Chat-Completions-shaped upstream (`targetFormat === FORMATS.OPENAI` — `reasoning_effort` is an OpenAI-shaped field, so a Claude- or Gemini-targeted request is unaffected). Opt in via `X-OmniRoute-Effort: auto` or a model's `defaultReasoningEffort: "auto"` (now a valid `ModelSpec` value); any explicit client reasoning field always wins ([#13448](https://github.com/diegosouzapw/OmniRoute/pull/13448)) diff --git a/changelog.d/fixes/13617-devin-cli-key-validation-fallback.md b/changelog.d/fixes/13617-devin-cli-key-validation-fallback.md new file mode 100644 index 00000000000..03d7fd58f27 --- /dev/null +++ b/changelog.d/fixes/13617-devin-cli-key-validation-fallback.md @@ -0,0 +1 @@ +- **fix(devin):** Fall back to the CLI probe (`devin acp --agent-type summarizer`) when the connection-test HTTP API rejects a CLI-format key, since routing authenticates against the local Devin CLI, not `api.devin.ai` ([#13617](https://github.com/diegosouzapw/OmniRoute/pull/13617)) diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index 162409c5a6f..55b410b2dc5 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -1702,11 +1702,6 @@ "count": 1 } }, - "src/lib/providers/validation/webProvidersB.ts": { - "@typescript-eslint/no-unused-vars": { - "count": 1 - } - }, "src/lib/quota/quotaAdapters.ts": { "@typescript-eslint/no-unused-vars": { "count": 1 diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 307648fe89a..31dd14b671b 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -451,6 +451,7 @@ "_rebaseline_2026_09_15_13440_daily_reset_tz": "#13440 rework: open-sse/services/accountFallback.ts 2469->2493 (+24): +6 for the operator-clock-first branch in checkFallbackError non-TPD daily quota (nextConfiguredResetMs leaf lives in dailyQuotaReset.ts, under cap) and +18 from the mandatory lint-staged Prettier pass over pre-existing unformatted lines of the touched file (no logic). executeTargetAttempt.ts 1212->1215 and roundRobinCombo.ts 1205->1208 (+3 each): one import plus the rotation/dailyReset arguments at the existing checkFallbackError call site; the lookup itself is the new comboDailyResetClock.ts leaf (under cap). Covered by tests/unit/daily-reset-tz-threading.test.ts.", "_rebaseline_2026_09_15_13672_retry_after_provenance": "#13672 rework (opt-in RETRY_AFTER_PROVENANCE_ENABLED): open-sse/services/combo/executeTargetAttempt.ts 1212->1220 (+8) and roundRobinCombo.ts 1205->1210 (+5) at the existing drain-path clone/parse block: capture the already-read body text, log an unreadable hint (debug for a non-JSON page, warn for a failed clone) instead of an empty catch, and one flag-gated prose fallback line; the import grows by the two helpers. Parsing, flag read and the Retry-After/provenance logic live in open-sse/utils/error.ts (under cap). Covered by tests/unit/retry-after-provenance.test.ts (flag off and on).", "_rebaseline_2026_09_15_13439_protected_priority_stop_status": "#13439 rework (opt-in PROTECTED_PRIORITY_INFRA_502_ENABLED): open-sse/services/combo/executeTargetAttempt.ts 1212->1217 (+5): two import lines for the new protectedPriorityStopStatus.ts leaf (where the provably-non-quota cause list and the flag read live) and the predictive_ttft cause argument at the existing stopProtectedPriorityTarget call, which Prettier splits over three lines. Covered by tests/unit/combo/protected-priority-stop-status-13439.test.ts (every stop cause, flag off and on).", + "_rebaseline_2026_09_16_13448_adaptive_effort_targetformat_gate": "#13448 rework: open-sse/handlers/chatCore.ts 6142->6156 (+14, PR's own growth: the X-OmniRoute-Effort header capture near THINKING_MARKER_HEADER and the wireAdaptiveEffort(translatedBody, {...}) call site right after applyDefaultReasoningEffort, plus this rework's +1 targetFormat argument at that same call). The targetFormat gate itself (ctx.targetFormat !== FORMATS.OPENAI short-circuit) lives in the non-frozen open-sse/handlers/chatCore/adaptiveEffortWiring.ts leaf, not here -- irreducible call-site wiring at the existing post-translation reasoning-normalization chokepoint. Covered by tests/unit/adaptive-effort-wiring.test.ts (targetFormat gate, red-on-tip) and tests/unit/adaptive-effort-model-default-13448.test.ts.", "_rebaseline_pr1043_minimax_tts": "Upstream port decolua/9router#1043 (toanalien) own growth: audioSpeech.ts 965->1061 (+96). Adds MiniMax T2A v2 TTS dispatch (handleMinimaxSpeech + hexToBytes helper) — provider entry was already in audioRegistry (format: minimax-tts) but no handler existed, falling through to the OpenAI-compatible default that fails (T2A has custom shape + hex-encoded audio + base_resp envelope). New branch sits next to the other inline provider branches (xiaomi-mimo, coqui, tortoise, aws-polly) — extracting would just create indirection. Covered by tests/unit/minimax-tts-1043.test.ts (3 tests, GREEN: success, base_resp error, invalid-hex).", "_rebaseline_pr4592_exclude_exhausted_auto": "Reconcile #4592 already-merged growth: combo.ts 2991->3036 (+45, terminal-status quota-cutoff exclusion in buildAutoCandidates + opt-in gate). Fast-gate PR->release does not run check:file-size.", "open-sse/executors/antigravity.ts": 1665, @@ -459,7 +460,7 @@ "open-sse/executors/codex.ts": 1505, "open-sse/executors/cursor.ts": 1808, "open-sse/executors/muse-spark-web.ts": 1405, - "open-sse/handlers/chatCore.ts": 6146, + "open-sse/handlers/chatCore.ts": 6156, "open-sse/handlers/imageGeneration.ts": 3293, "open-sse/handlers/search.ts": 1789, "open-sse/mcp-server/schemas/tools.ts": 1621, diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index aea0ffb0a92..3f1fbf36387 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -104,7 +104,8 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | `OMNIROUTE_WAL_TRUNCATE_INTERVAL_MS` | `21600000` (6h) | `src/lib/db/walMaintenance.ts` | Override the periodic `wal_checkpoint(TRUNCATE)` interval (ms). Auto-checkpoint never shrinks the WAL file itself, and a long-running server never closes its DB. `0` disables. | | `OMNIROUTE_WAL_PASSIVE_INTERVAL_MS` | `300000` (5m) | `src/lib/db/walMaintenance.ts` | Override the frequent `wal_checkpoint(PASSIVE)` interval (ms). Keeps pending WAL frames small so the periodic TRUNCATE never copies a multi-GB backlog on the main thread. `0` disables. | | `OMNIROUTE_WAL_GUARD_MAX_MB` | `256` | `src/lib/db/walMaintenance.ts` | When a PASSIVE tick finds the WAL file above this size, escalate to `wal_checkpoint(TRUNCATE)` immediately instead of waiting for the slow tick. | -| `OMNIROUTE_VACUUM_MIN_DELETED_ROWS` | `1000` | `src/lib/db/cleanup.ts` | Post-cleanup VACUUM only runs when the cleanup deleted at least this many rows. `0` means always VACUUM when a cleanup freed any rows; `1` effectively disables the gate. VACUUM rewrites the entire database (multi-GB WAL + I/O burst on large DBs), so tiny cleanups skip it; the Storage page's scheduled VACUUM (default weekly, #4437) and manual VACUUM still reclaim space. | +| `OMNIROUTE_VACUUM_MIN_DELETED_ROWS` | `1000` | `src/lib/db/cleanup.ts` | Post-cleanup VACUUM runs when the cleanup deleted at least this many rows, OR when `OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB` below is met (whichever fires first). `0` means always VACUUM when a cleanup freed any rows; `1` effectively disables the row-count gate. VACUUM rewrites the entire database (multi-GB WAL + I/O burst on large DBs), so tiny cleanups skip it; the Storage page's scheduled VACUUM (default weekly, #4437) and manual VACUUM still reclaim space. | +| `OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB` | `100` | `src/lib/db/cleanup.ts` | Minimum reclaimable space (SQLite's own free-page count, in MB) that alone triggers the post-cleanup `VACUUM`, even when `OMNIROUTE_VACUUM_MIN_DELETED_ROWS` was not met -- a handful of oversized blob rows can free far more space than thousands of tiny rows. `VACUUM` is synchronous and blocks the entire process (15-20 min on a multi-GB database). `0` always vacuums after any deletion. | | `OMNIROUTE_PRESSURE_SELF_RESTART` | `false` | `open-sse/utils/resourcePressure.ts` | Set to `1`/`true`/`yes`/`on` to exit the process after critical resource pressure is sustained for `OMNIROUTE_PRESSURE_SELF_RESTART_AFTER_MS`, letting a supervisor (systemd `Restart=always`, Docker restart policy) bring back a clean process instead of serving 503s indefinitely. | | `OMNIROUTE_PRESSURE_SELF_RESTART_AFTER_MS` | `120000` (2m) | `open-sse/utils/resourcePressure.ts` | How long critical pressure must persist before the self-restart exit fires. | | `OMNIROUTE_SQLJS_WASM_PATH` | _(auto-detect)_ | `src/lib/db/adapters/sqljsAdapter.ts` | Explicit path (absolute or relative to cwd) to `sql-wasm.wasm` when using the `sql.js` WASM fallback adapter. Auto-detected via package dependencies and candidate layouts when unset. | diff --git a/docs/routing/AUTO-COMBO.md b/docs/routing/AUTO-COMBO.md index 2b406a7b2c3..e3970886403 100644 --- a/docs/routing/AUTO-COMBO.md +++ b/docs/routing/AUTO-COMBO.md @@ -252,11 +252,12 @@ combo's stored config. These apply only to the `auto` strategy and only for the that carries them; the combo's saved `modePack`/`budgetCap`/`budgetFallback` are used when the header is absent. -| Header | Accepts | Effect | -| :---------------------------- | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `X-OmniRoute-Mode` | a preset alias (`fast`, `balanced`, `quality`, `cheap`, `reliable`, `offline`) or a raw pack name (`ship-fast`, `cost-saver`, `quality-first`, `offline-friendly`, `reliability-first`) | Overrides the scoring weights for this request. `balanced`/`default` force the default weights (no pack). Unknown values are ignored (config preserved). | -| `X-OmniRoute-Budget` | a positive number (max USD per request) | Hard cost ceiling: candidates whose estimated cost exceeds it are filtered before selection. What happens when **every** candidate exceeds it is controlled by `X-OmniRoute-Budget-Fallback` below. | -| `X-OmniRoute-Budget-Fallback` | `cheapest` (default, aliases: `cheapest-viable`, `soft`) or `strict` (aliases: `block`, `hard`) | `cheapest`: falls back to the globally cheapest candidate even though it still exceeds the cap (legacy behavior). `strict`: refuses to select — the request fails fast with `HTTP 402` instead of silently overspending. Unknown values are ignored. | +| Header | Accepts | Effect | +| :---------------------------- | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `X-OmniRoute-Mode` | a preset alias (`fast`, `balanced`, `quality`, `cheap`, `reliable`, `offline`) or a raw pack name (`ship-fast`, `cost-saver`, `quality-first`, `offline-friendly`, `reliability-first`) | Overrides the scoring weights for this request. `balanced`/`default` force the default weights (no pack). Unknown values are ignored (config preserved). | +| `X-OmniRoute-Budget` | a positive number (max USD per request) | Hard cost ceiling: candidates whose estimated cost exceeds it are filtered before selection. What happens when **every** candidate exceeds it is controlled by `X-OmniRoute-Budget-Fallback` below. | +| `X-OmniRoute-Budget-Fallback` | `cheapest` (default, aliases: `cheapest-viable`, `soft`) or `strict` (aliases: `block`, `hard`) | `cheapest`: falls back to the globally cheapest candidate even though it still exceeds the cap (legacy behavior). `strict`: refuses to select — the request fails fast with `HTTP 402` instead of silently overspending. Unknown values are ignored. | +| `X-OmniRoute-Effort` | `auto` (other values reserved) | Adaptive thinking budget: when the request carries **no** reasoning field of any shape (`reasoning_effort`, `reasoning`, `thinking`), the gateway resolves `auto` to `low`/`medium`/`high` from deterministic request-shape signals (last-user-message length, context size up to the last user message, prior tool results, tool-loop depth). Signals are scoped to the current turn — everything after the last user message is ignored — so every request in a tool loop resolves to the same level (stateless per-turn pin, no session state, no mid-loop escalation that would break upstream prompt-cache prefixes). An explicit client reasoning field always wins. Scoped to requests whose upstream dispatch resolves to the OpenAI Chat Completions shape (`targetFormat === FORMATS.OPENAI`) — `reasoning_effort` is an OpenAI-shaped field, so the header is a no-op on a Claude- or Gemini-targeted request (see `open-sse/handlers/chatCore/adaptiveEffortWiring.ts`). | ```bash # Force the fastest profile, cap this request at $0.05, and hard-block instead of overspending diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 7ef9ef55417..45c0c134dbb 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -170,6 +170,7 @@ import { import { shouldUseMidConversationSystem } from "../executors/claudeIdentity.ts"; import { normalizeClaudeHaikuConstraints } from "../services/claudeHaikuConstraints.ts"; import { applyDefaultReasoningEffort } from "../services/defaultReasoningEffort.ts"; +import { wireAdaptiveEffort } from "./chatCore/adaptiveEffortWiring.ts"; import { echoModelInObject } from "../services/responseModelEcho.ts"; import { stripGpt5SamplingWhenReasoning, @@ -2718,6 +2719,11 @@ export async function handleChatCore({ (modelInfo as { defaultThinkingEffort?: string })?.defaultThinkingEffort ); } + translatedBody = wireAdaptiveEffort(translatedBody, { + rawBody: body, + clientRawRequest, + targetFormat, + }); } // Xiaomi MiMo controls reasoning ONLY via `thinking:{type:"enabled"|"disabled"}` and diff --git a/open-sse/handlers/chatCore/adaptiveEffortWiring.ts b/open-sse/handlers/chatCore/adaptiveEffortWiring.ts new file mode 100644 index 00000000000..c891f7b15af --- /dev/null +++ b/open-sse/handlers/chatCore/adaptiveEffortWiring.ts @@ -0,0 +1,72 @@ +// Adaptive reasoning-effort wiring (#13448), extracted from chatCore.ts so the +// frozen main file does not grow (file-size gate: chatCore.ts cannot grow). +// Semantics live in open-sse/services/adaptiveEffort.ts; this module only +// adapts the chatCore call-site context (headers, raw body, translated body). +// +// Runs AFTER applyDefaultReasoningEffort so its explicit-value precedence and +// alias-suffix priority are preserved; operates on the pre-translation body so +// source-format differences are handled by the existing translators. +import { + applyAdaptiveEffort, + hasExplicitReasoningField, + isAdaptiveEffort, + type ChatMessageLike, +} from "../../services/adaptiveEffort.ts"; +import { FORMATS } from "../../translator/formats.ts"; +import { getHeaderValueCaseInsensitive } from "./headers.ts"; + +export interface AdaptiveEffortContext { + /** Raw (pre-translation) request body, for turn-scoped request-shape signals. */ + rawBody: { messages?: ChatMessageLike[] | undefined } | undefined; + /** Incoming client request, used to read the x-omniroute-effort header. */ + clientRawRequest?: { headers?: unknown } | undefined; + /** Explicit header value, if already extracted by the caller. */ + headerEffort?: string | null | undefined; + /** + * Resolved upstream dispatch format (chatCore.ts's `targetFormat`). `reasoning_effort` + * is an OpenAI Chat-Completions-shaped field: on any other target it either does + * nothing (Claude/Gemini executors read `thinking`/`reasoning.effort` instead and + * never look at it) or, worse, reaches an upstream that rejects unrecognized + * top-level parameters (e.g. Anthropic's Messages API 400s on one). Every other + * reasoning-shape normalization in chatCore.ts (applyDefaultReasoningEffort, + * promoteStrayReasoningEffort for the Responses same-format lane) is scoped the + * same way — wiring must match, or an operator's `X-OmniRoute-Effort: auto` header + * on a Claude/Gemini-targeted request would silently no-op or break the request. + */ + targetFormat: string | undefined; +} + +/** + * Resolve "auto" reasoning effort to a concrete level when the request opted in + * (header or ModelSpec.defaultReasoningEffort === "auto") and carries no explicit + * reasoning field. Returns `body` unchanged (same reference) otherwise. + * + * Scoped to `FORMATS.OPENAI` dispatch — see {@link AdaptiveEffortContext.targetFormat}. + */ +export function wireAdaptiveEffort>( + body: T, + ctx: AdaptiveEffortContext +): T { + if (ctx.targetFormat !== FORMATS.OPENAI) return body; + // Lever: applyDefaultReasoningEffort may have just injected the literal + // "auto" from ModelSpec.defaultReasoningEffort — that is an opt-in marker, + // not a wire value, so it must NOT count as an explicit client field (it + // would otherwise short-circuit the guard below and ship "auto" upstream). + const modelDefaultAuto = isAdaptiveEffort(body.reasoning_effort); + if (!modelDefaultAuto && hasExplicitReasoningField(body)) return body; + const headerEffort = + ctx.headerEffort !== undefined + ? ctx.headerEffort + : getHeaderValueCaseInsensitive( + ctx.clientRawRequest?.headers as Record | Headers | null | undefined, + "x-omniroute-effort" + ); + if (!modelDefaultAuto && !isAdaptiveEffort(headerEffort)) return body; + const stripped = modelDefaultAuto ? { ...body } : body; + if (modelDefaultAuto) delete (stripped as Record).reasoning_effort; + return applyAdaptiveEffort(stripped, { + messages: ctx.rawBody?.messages, + headerEffort: headerEffort ?? null, + modelDefaultEffort: modelDefaultAuto ? "auto" : null, + }) as T; +} diff --git a/open-sse/services/adaptiveEffort.ts b/open-sse/services/adaptiveEffort.ts new file mode 100644 index 00000000000..2daf2d2a8c2 --- /dev/null +++ b/open-sse/services/adaptiveEffort.ts @@ -0,0 +1,137 @@ +// Adaptive reasoning effort — the OmniRoute-side counterpart of Hermes' +// `effort: "auto"` (NousResearch/hermes-agent#109044). One implementation at +// the gateway covers every harness (Claude Code, Cursor, Codex, opencode, +// Hermes) because the full request body passes through here before any +// provider translation. +// +// Semantics: +// - Effort is resolved from deterministic request-shape signals only — no LLM +// call, no judgment gate. Three bands (low / medium / high) gate a thinking +// budget, not a model-routing decision. +// - STATELESS PER-TURN PIN: signals are computed ONLY from the last user +// message and everything BEFORE it. Tool results after the last user +// message are ignored, so every request of the same user turn — including +// mid-tool-loop requests — resolves to the SAME level. This reproduces +// Hermes' stateful per-turn pin deterministically, without stored state, +// and never escalates mid-loop (which would change the reasoning config +// between requests and cost a cold prompt-cache prefix write on +// cache-sensitive upstreams). +// +// Priority (highest first), all off-by-default: +// 1. Explicit client reasoning field of any shape — always wins; no-op. +// 2. `X-OmniRoute-Effort: auto` request header (per-request opt-in; mirrors +// the #6023/#6024/#6025 `X-OmniRoute-Mode`/`-Budget` controls pattern). +// 3. `ModelSpec.defaultReasoningEffort: "auto"` (per-model opt-in, #6879). +import { estimateMessageTokens } from "./specificityRules"; + +export const ADAPTIVE_EFFORT = "auto"; + +// Thresholds mirror the Hermes resolver (agent/reasoning_effort.py): three +// coarse deterministic bands. A near-miss costs a slightly over/under-thought +// answer, not a wrong route, so they stay coarse until call-log data says +// otherwise. +const TRIVIAL_USER_CHARS = 160; +const TRIVIAL_CTX_TOKENS = 4000; +const HEAVY_CTX_TOKENS = 60000; +const HEAVY_TOOL_RESULTS = 6; +const HEAVY_USER_CHARS = 4000; + +export type ChatMessageLike = { role?: unknown; content?: unknown }; +type EffortLevel = "low" | "medium" | "high"; + +function isString(v: unknown): v is string { + return typeof v === "string"; +} + +function lastUserMessageIndex(messages: ChatMessageLike[]): number { + let last = -1; + for (let i = 0; i < messages.length; i++) { + if (messages[i]?.role === "user") last = i; + } + return last; +} + +function messageTextChars(content: unknown): number { + if (isString(content)) return content.length; + if (Array.isArray(content)) { + let sum = 0; + for (const part of content) { + const text = (part as { text?: unknown })?.text; + if (isString(text)) sum += text.length; + } + return sum; + } + return 0; +} + +function countRole(messages: ChatMessageLike[], role: string, from: number, to: number): number { + let n = 0; + for (let i = from; i < to; i++) { + if (messages[i]?.role === role) n++; + } + return n; +} + +function toolLoopDepthAfter(messages: ChatMessageLike[], boundary: number): number { + let n = 0; + for (let i = boundary + 1; i < messages.length; i++) { + const msg = messages[i]; + if (msg?.role === "assistant" && (msg as { tool_calls?: unknown }).tool_calls) n++; + } + return n; +} + +export function resolveAdaptiveEffort(messages: ChatMessageLike[] | undefined | null): EffortLevel { + const msgs = Array.isArray(messages) ? messages : []; + if (msgs.length === 0) return "medium"; // no signals at all → balanced band, never cheap-by-default + const boundary = lastUserMessageIndex(msgs); + const upToTurn = boundary >= 0 ? msgs.slice(0, boundary + 1) : msgs; + const userChars = boundary >= 0 ? messageTextChars(msgs[boundary].content) : 0; + const estCtxTokens = estimateMessageTokens(upToTurn as Array<{ content?: unknown }>); + const toolResults = boundary >= 0 ? countRole(msgs, "tool", 0, boundary) : 0; + const turnDepth = boundary >= 0 ? toolLoopDepthAfter(msgs, boundary) + 1 : 1; + + const trivial = + userChars <= TRIVIAL_USER_CHARS && + estCtxTokens <= TRIVIAL_CTX_TOKENS && + toolResults === 0 && + turnDepth <= 1; + if (trivial) return "low"; + const heavy = + estCtxTokens >= HEAVY_CTX_TOKENS || + toolResults >= HEAVY_TOOL_RESULTS || + userChars >= HEAVY_USER_CHARS; + if (heavy) return "high"; + return "medium"; +} + +export function isAdaptiveEffort(value: unknown): boolean { + return isString(value) && value.trim().toLowerCase() === ADAPTIVE_EFFORT; +} + +export function hasExplicitReasoningField(body: Record): boolean { + return ( + body.reasoning_effort !== undefined || + body.reasoning !== undefined || + body.thinking !== undefined + ); +} + +export function applyAdaptiveEffort>( + body: T, + opts: { + messages?: ChatMessageLike[] | undefined | null; + headerEffort?: unknown; + modelDefaultEffort?: string | null; + } +): T { + if (!body || typeof body !== "object") return body; + if (hasExplicitReasoningField(body)) return body; + const headerAuto = isAdaptiveEffort(opts.headerEffort); + const defaultAuto = opts.modelDefaultEffort != null && isAdaptiveEffort(opts.modelDefaultEffort); + if (!headerAuto && !defaultAuto) return body; + const level = resolveAdaptiveEffort( + opts.messages ?? (body.messages as ChatMessageLike[] | undefined) + ); + return { ...body, reasoning_effort: level }; +} diff --git a/src/lib/db/cleanup.ts b/src/lib/db/cleanup.ts index 19b92dac175..1ca7c148690 100644 --- a/src/lib/db/cleanup.ts +++ b/src/lib/db/cleanup.ts @@ -922,23 +922,85 @@ export function shouldVacuumAfterCleanup( return totalDeleted > 0 && totalDeleted >= minRows; } +const DEFAULT_VACUUM_MIN_RECLAIMABLE_BYTES = 100 * 1024 * 1024; // 100 MB + +export function getVacuumMinReclaimableBytes(): number { + const raw = process.env.OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB; + if (!raw) return DEFAULT_VACUUM_MIN_RECLAIMABLE_BYTES; + const parsed = Number.parseInt(raw, 10); + return Number.isInteger(parsed) && parsed >= 0 + ? parsed * 1024 * 1024 + : DEFAULT_VACUUM_MIN_RECLAIMABLE_BYTES; +} + /** - * Runs the post-cleanup VACUUM only when the cleanup freed enough rows to - * justify a full-database rewrite. Returns true when VACUUM ran. + * `db.exec("VACUUM")` is synchronous and blocks the entire process -- on a + * multi-GB database that can freeze all HTTP traffic for 15-20 minutes, even + * when the cleanup that triggered it only freed a handful of rows. Reclaimable + * space (SQLite's own free-page count, not row count) is what actually + * determines whether that multi-minute freeze is worth it: a few oversized + * batch_item_checkpoints rows can free more than thousands of tiny audit-log + * rows. Observed live: a routine cleanup that freed 2-6 rows re-triggered a + * full VACUUM on every restart regardless. + */ +export function getReclaimableBytes(db: ReturnType): number { + const freelist = db.pragma("freelist_count", { simple: true }) as number; + const pageSize = db.pragma("page_size", { simple: true }) as number; + return freelist * pageSize; +} + +/** + * Runs the post-cleanup VACUUM when EITHER the row-count threshold + * (OMNIROUTE_VACUUM_MIN_DELETED_ROWS, default 1000 -- see shouldVacuumAfterCleanup) + * OR the reclaimable-bytes threshold (OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB, default + * 100 MB -- see getReclaimableBytes) is met. The two signals are additive on + * purpose: row count alone misses the case where a handful of oversized blob + * rows (e.g. batch_item_checkpoints) free far more space than thousands of tiny + * audit-log rows would, while reclaimable bytes alone would never fire for a + * cleanup that deletes many small rows without freeing much space yet still + * crosses the operator's row-count comfort threshold. `getReclaimable` is + * optional and defaults to skipping the bytes check entirely, so existing + * callers/tests that only care about the row-count gate keep their exact + * behavior unchanged. Returns true when VACUUM ran. */ export async function vacuumAfterCleanup( totalDeleted: number, exec: (sql: string) => void, log: (message: string) => void = (m) => console.log(m), - logError: (message: string, error: unknown) => void = (m, e) => console.error(m, e) + logError: (message: string, error: unknown) => void = (m, e) => console.error(m, e), + getReclaimable?: () => number ): Promise { if (totalDeleted <= 0) return false; const minRows = getVacuumMinDeletedRows(); - if (!shouldVacuumAfterCleanup(totalDeleted, minRows)) { - log(`[Cleanup] Freed ${totalDeleted} rows; skipping VACUUM (below ${minRows}-row threshold).`); + const rowThresholdMet = shouldVacuumAfterCleanup(totalDeleted, minRows); + + let reclaimable = 0; + let reclaimableThresholdMet = false; + const minBytes = getVacuumMinReclaimableBytes(); + if (typeof getReclaimable === "function") { + try { + reclaimable = getReclaimable(); + reclaimableThresholdMet = reclaimable >= minBytes; + } catch (err) { + logError("[Cleanup] Failed to read reclaimable space:", err); + } + } + + if (!rowThresholdMet && !reclaimableThresholdMet) { + log( + `[Cleanup] Freed ${totalDeleted} rows; skipping VACUUM (below ${minRows}-row threshold` + + (typeof getReclaimable === "function" + ? ` and below ${(minBytes / (1024 * 1024)).toFixed(0)} MB reclaimable, only ${(reclaimable / (1024 * 1024)).toFixed(1)} MB)` + : ")") + ); return false; } - log(`[Cleanup] Running VACUUM to reclaim ${totalDeleted} freed rows...`); + log( + `[Cleanup] Running VACUUM (${totalDeleted} rows freed` + + (typeof getReclaimable === "function" + ? `, ${(reclaimable / (1024 * 1024)).toFixed(1)} MB reclaimable)...` + : ")...") + ); try { exec("VACUUM"); log("[Cleanup] VACUUM completed after cleanup."); @@ -948,14 +1010,16 @@ export async function vacuumAfterCleanup( return false; } } - /** * Start the background cleanup scheduler. Runs cleanup on startup - * and then every 6 hours. Runs VACUUM after deletes to reclaim disk space. + * and then every 6 hours. VACUUMs after deletes only when the cleanup freed + * enough rows (OMNIROUTE_VACUUM_MIN_DELETED_ROWS, default 1000) OR freed + * enough reclaimable space (OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB, default + * 100 MB) to justify a full-database rewrite -- see vacuumAfterCleanup(). * - * Without this, tables grow unboundedly (compression_analytics 600K+ rows, - * usage_history 250K+ rows) causing 1.4GB+ SQLite files and 3-8GB RSS - * from better-sqlite3 memory mapping. + * Without the cleanup itself, tables grow unboundedly (compression_analytics + * 600K+ rows, usage_history 250K+ rows) causing 1.4GB+ SQLite files and + * 3-8GB RSS from better-sqlite3 memory mapping. */ export function startCleanupScheduler(): void { if (_cleanupSchedulerTimer) return; @@ -968,7 +1032,13 @@ export function startCleanupScheduler(): void { const totalDeleted = result.totalDeleted + proxyResult.deleted; if (totalDeleted > 0) { console.log(`[Cleanup] Startup cleanup freed ${totalDeleted} rows.`); - await vacuumAfterCleanup(totalDeleted, (sql) => getDbInstance().exec(sql)); + await vacuumAfterCleanup( + totalDeleted, + (sql) => getDbInstance().exec(sql), + undefined, + undefined, + () => getReclaimableBytes(getDbInstance()) + ); } } catch (err) { console.error("[Cleanup] Startup cleanup failed:", err); @@ -983,7 +1053,13 @@ export function startCleanupScheduler(): void { const totalDeleted = result.totalDeleted + proxyResult.deleted; if (totalDeleted > 0) { console.log(`[Cleanup] Periodic cleanup freed ${totalDeleted} rows.`); - await vacuumAfterCleanup(totalDeleted, (sql) => getDbInstance().exec(sql)); + await vacuumAfterCleanup( + totalDeleted, + (sql) => getDbInstance().exec(sql), + undefined, + undefined, + () => getReclaimableBytes(getDbInstance()) + ); } } catch (err) { console.error("[Cleanup] Periodic cleanup failed:", err); diff --git a/src/lib/providers/validation/webProvidersB.ts b/src/lib/providers/validation/webProvidersB.ts index d69a5b4e621..e2a6400b19c 100644 --- a/src/lib/providers/validation/webProvidersB.ts +++ b/src/lib/providers/validation/webProvidersB.ts @@ -2,6 +2,7 @@ // copilot-web, t3-web, jules, devin (cloud-agent), inner-ai. Extracted from validation.ts (god-file // decomposition) — top-level functions with no dispatcher-state captures; behavior is byte-identical // to the inline defs. +import { spawn } from "child_process"; import { applyCustomUserAgent } from "./headers"; import { isSecurityBlockError, @@ -514,12 +515,51 @@ export async function validateJulesProvider({ apiKey }: { apiKey: string }) { } } +/** + * #devin-cli-key: fallback validator for CLI-format Devin keys. + * + * The devin provider's actual routing path (open-sse/executors/devin-cli.ts) + * shells out to the Devin CLI binary and passes the connection's apiKey as + * WINDSURF_API_KEY — never touching api.devin.ai. CLI keys (apk_user_…) are + * rejected by the HTTP API, so a 401 from the HTTP probe is NOT evidence the + * connection is broken. This runs the same probe the executor uses: + * `devin acp --agent-type summarizer` with the key in the environment + * (`devin models list` does NOT honor WINDSURF_API_KEY). Exit 0 = key works. + */ +async function validateDevinCliKeyFallback( + apiKey: unknown +): Promise<{ valid: boolean; error: string | null }> { + const bin = process.env.CLI_DEVIN_BIN?.trim() || "devin"; + return new Promise((resolve) => { + try { + const child = spawn(bin, ["acp", "--agent-type", "summarizer"], { + env: { ...process.env, WINDSURF_API_KEY: String(apiKey || "") }, + stdio: ["ignore", "pipe", "pipe"], + timeout: 30_000, + }); + child.on("error", () => + resolve({ valid: false, error: "Devin CLI not available for fallback validation" }) + ); + child.on("close", (code) => { + if (code === 0) resolve({ valid: true, error: null }); + else resolve({ valid: false, error: `Devin CLI key check failed (exit ${code})` }); + }); + } catch { + resolve({ valid: false, error: "Devin CLI fallback spawn failed" }); + } + }); +} + /** * Devin cloud-agent (Cognition) — GET /v1/sessions with Bearer auth * (see docs.devin.ai/api-reference/sessions/list-sessions). Distinct from the * "devin-cli" LLM provider (ACP), which is already wired via providerRegistry. */ -export async function validateDevinCloudAgentProvider({ apiKey }: { apiKey: string }) { +export async function validateDevinCloudAgentProvider({ + apiKey, +}: { + apiKey: string; +}): Promise<{ valid: boolean; error: string | null; warning?: string }> { try { const response = await validationWrite("https://api.devin.ai/v1/sessions?limit=1", { method: "GET", @@ -529,6 +569,18 @@ export async function validateDevinCloudAgentProvider({ apiKey }: { apiKey: stri }); if (response.status === 401 || response.status === 403) { + // #devin-cli-key: CLI-format keys (apk_user_…) are rejected by the HTTP API + // but are exactly what the devin-cli executor authenticates with (via + // WINDSURF_API_KEY). Fall back to probing the CLI itself — the real + // routing path — before declaring the key invalid. + const cliCheck = await validateDevinCliKeyFallback(apiKey); + if (cliCheck.valid) { + return { + valid: true, + error: null, + warning: "HTTP API rejected this key; validated via Devin CLI instead", + }; + } return { valid: false, error: "Invalid API key" }; } @@ -597,7 +649,7 @@ export async function validateNotionWebProvider({ apiKey, providerSpecificData = } } -export async function validateInnerAiProvider({ apiKey, providerSpecificData = {} }: any) { +export async function validateInnerAiProvider({ apiKey }: any) { try { const raw = typeof apiKey === "string" ? apiKey.trim() : ""; if (!raw) { diff --git a/src/shared/constants/modelSpecs.ts b/src/shared/constants/modelSpecs.ts index f623bc048de..57c19428f62 100644 --- a/src/shared/constants/modelSpecs.ts +++ b/src/shared/constants/modelSpecs.ts @@ -48,7 +48,17 @@ export interface ModelSpec { // operator strip-by-default a thinks-by-default model (measured: gemini-flash-lite // burns ~277 reasoning tokens on a plain request; `reasoning_effort:"none"` → 0) // without patching every client. See open-sse/services/defaultReasoningEffort.ts. - defaultReasoningEffort?: "none" | "low" | "medium" | "high"; + // + // `"auto"` (#13448) is the per-model opt-in into adaptive reasoning effort: the + // literal value is injected here exactly like any other level, then + // chatCore/adaptiveEffortWiring.ts's wireAdaptiveEffort() recognizes it as an + // opt-in marker (never forwarded upstream verbatim) and resolves it to a + // concrete low/medium/high from the turn's request-shape signals. Without + // "auto" in this union, no operator could configure the per-model opt-in + // through the typed catalog at all -- open-sse/services/adaptiveEffort.ts's + // priority #3 and the wiring's modelDefaultAuto branch were unreachable + // except by a test constructing the body literal directly. + defaultReasoningEffort?: "none" | "low" | "medium" | "high" | "auto"; } const BEDROCK_CLAUDE_ALIASES = (...modelIds: string[]) => [ diff --git a/tests/unit/adaptive-effort-model-default-13448.test.ts b/tests/unit/adaptive-effort-model-default-13448.test.ts new file mode 100644 index 00000000000..b7c5e51151a --- /dev/null +++ b/tests/unit/adaptive-effort-model-default-13448.test.ts @@ -0,0 +1,97 @@ +/** + * #13448 rework — the per-model opt-in path (`ModelSpec.defaultReasoningEffort: + * "auto"`) was unreachable through the real, type-checked catalog: the field's + * type union was `"none" | "low" | "medium" | "high"`, so no operator config in + * providerRegistry.ts (or a fixture like this one) could ever assign `"auto"` + * without a type error. `open-sse/services/adaptiveEffort.ts`'s priority #3 and + * `adaptiveEffortWiring.ts`'s `modelDefaultAuto` branch were only ever exercised + * by tests that constructed the post-injection body literal directly + * (`{ reasoning_effort: "auto" }`), bypassing the type entirely. + * + * These tests exercise the REAL two-function pipeline end to end, starting + * from a typed `MODEL_SPECS` entry (no `as any`, no literal shortcut): + * MODEL_SPECS[id].defaultReasoningEffort === "auto" + * -> applyDefaultReasoningEffort() injects the literal "auto" + * -> wireAdaptiveEffort() recognizes it as an opt-in marker and resolves + * it to a concrete low/medium/high (OpenAI dispatch only, #13448 rework + * -- see adaptive-effort-wiring.test.ts for the targetFormat gate). + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import { applyDefaultReasoningEffort } from "../../open-sse/services/defaultReasoningEffort.ts"; +import { wireAdaptiveEffort } from "../../open-sse/handlers/chatCore/adaptiveEffortWiring.ts"; +import { FORMATS } from "../../open-sse/translator/formats.ts"; +import { MODEL_SPECS, type ModelSpec } from "../../src/shared/constants/modelSpecs.ts"; + +const FIXTURE_MODEL_ID = "__test_13448_model_default_auto__"; + +// Typed assignment through the real ModelSpec union -- this line alone would +// be a TypeScript error on the pristine tip (`Type '"auto"' is not assignable +// to type '"none" | "low" | "medium" | "high"'`), which is exactly the +// unreachability defect: verify with `npm run -s typecheck:core`. +const fixtureSpec: ModelSpec = { defaultReasoningEffort: "auto" }; + +test.before(() => { + MODEL_SPECS[FIXTURE_MODEL_ID] = fixtureSpec; +}); + +test.after(() => { + delete MODEL_SPECS[FIXTURE_MODEL_ID]; +}); + +const HEAVY = "x".repeat(20000); +const trivialMsgs = [{ role: "user", content: "list the files" }]; +const heavyMsgs = [{ role: "user", content: HEAVY }]; + +test("a real ModelSpec.defaultReasoningEffort:'auto' entry is injected as the literal marker", () => { + const body = { model: FIXTURE_MODEL_ID, messages: [] }; + const result = applyDefaultReasoningEffort(body, FIXTURE_MODEL_ID); + assert.equal(result.reasoning_effort, "auto"); +}); + +test("end to end: the typed model-default 'auto' resolves to a concrete low level on OpenAI dispatch", () => { + const rawBody = { model: FIXTURE_MODEL_ID, messages: trivialMsgs }; + const afterDefault = applyDefaultReasoningEffort(rawBody, FIXTURE_MODEL_ID); + const wired = wireAdaptiveEffort(afterDefault, { + rawBody, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.notEqual(wired.reasoning_effort, "auto"); + assert.equal(wired.reasoning_effort, "low"); +}); + +test("end to end: the typed model-default 'auto' resolves to 'high' on a heavy turn", () => { + const rawBody = { model: FIXTURE_MODEL_ID, messages: heavyMsgs }; + const afterDefault = applyDefaultReasoningEffort(rawBody, FIXTURE_MODEL_ID); + const wired = wireAdaptiveEffort(afterDefault, { + rawBody, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(wired.reasoning_effort, "high"); +}); + +test("end to end: an explicit client reasoning_effort still wins over the typed model default", () => { + const rawBody = { model: FIXTURE_MODEL_ID, messages: heavyMsgs, reasoning_effort: "medium" }; + const afterDefault = applyDefaultReasoningEffort(rawBody, FIXTURE_MODEL_ID); + assert.equal(afterDefault.reasoning_effort, "medium", "no-op: explicit field already present"); + const wired = wireAdaptiveEffort(afterDefault, { + rawBody, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(wired.reasoning_effort, "medium"); +}); + +test("end to end: on a non-OpenAI target the typed model-default marker is left as literal 'auto' (never resolved, never forwarded resolved)", () => { + const rawBody = { model: FIXTURE_MODEL_ID, messages: heavyMsgs }; + const afterDefault = applyDefaultReasoningEffort(rawBody, FIXTURE_MODEL_ID); + const wired = wireAdaptiveEffort(afterDefault, { + rawBody, + headerEffort: null, + targetFormat: FORMATS.CLAUDE, + }); + assert.equal(wired, afterDefault, "same reference: wireAdaptiveEffort must no-op off OpenAI"); + assert.equal(wired.reasoning_effort, "auto"); +}); diff --git a/tests/unit/adaptive-effort-wiring.test.ts b/tests/unit/adaptive-effort-wiring.test.ts new file mode 100644 index 00000000000..fe7fa654b2e --- /dev/null +++ b/tests/unit/adaptive-effort-wiring.test.ts @@ -0,0 +1,225 @@ +// chatCore adaptive-effort wiring tests (#13448). +// +// The wiring module is the chatCore call-site adapter extracted from +// chatCore.ts (file-size gate: that file cannot grow). The service-level +// tests in adaptive-effort.test.ts cover resolution semantics; THESE tests +// cover the adapter's own decisions, which are invisible to the service: +// - explicit client reasoning fields are never overwritten (precedence), +// - the literal "auto" injected by ModelSpec.defaultReasoningEffort is an +// opt-in marker and must be stripped before resolution, not sent upstream, +// - the x-omniroute-effort header opts in independently (read inside the +// module from `clientRawRequest.headers`, or passed pre-extracted), +// - a non-opted-in body is returned untouched (same reference), +// - the whole wiring is scoped to OpenAI Chat-Completions dispatch. +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { wireAdaptiveEffort } from "@omniroute/open-sse/handlers/chatCore/adaptiveEffortWiring.ts"; +import { FORMATS } from "@omniroute/open-sse/translator/formats.ts"; + +const HEAVY = "x".repeat(20000); +const trivialMsgs = [{ role: "user", content: "list the files" }]; +const heavyMsgs = [{ role: "user", content: HEAVY }]; + +test("explicit reasoning_effort is never overwritten by adaptive wiring", () => { + const body = { model: "m", reasoning_effort: "low" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: "auto", + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out.reasoning_effort, "low"); +}); + +test("explicit reasoning object is never overwritten", () => { + const body = { model: "m", reasoning: { effort: "high" } }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: trivialMsgs }, + headerEffort: "auto", + targetFormat: FORMATS.OPENAI, + }); + assert.deepEqual(out.reasoning, { effort: "high" }); + assert.equal(out.reasoning_effort, undefined); +}); + +test("explicit thinking field is never overwritten", () => { + const body = { model: "m", thinking: { type: "enabled" } }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: "auto", + targetFormat: FORMATS.OPENAI, + }); + assert.deepEqual(out.thinking, { type: "enabled" }); + assert.equal(out.reasoning_effort, undefined); +}); + +test("model-default 'auto' marker is resolved, never sent upstream verbatim", () => { + const body = { model: "m", reasoning_effort: "auto" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: trivialMsgs }, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.notEqual(out.reasoning_effort, "auto"); + assert.equal(out.reasoning_effort, "low"); +}); + +test("model-default 'auto' resolves high on heavy turns", () => { + const body = { model: "m", reasoning_effort: "auto" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out.reasoning_effort, "high"); +}); + +test("header opt-in resolves from the raw (pre-translation) body messages", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: "auto", + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out.reasoning_effort, "high"); +}); + +test("no opt-in leaves the body untouched (same reference)", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out, body); + assert.equal(out.reasoning_effort, undefined); +}); + +test("missing rawBody does not throw", () => { + const body = { model: "m", reasoning_effort: "auto" }; + const out = wireAdaptiveEffort(body, { + rawBody: undefined, + headerEffort: null, + targetFormat: FORMATS.OPENAI, + }); + assert.ok(["low", "medium", "high"].includes(out.reasoning_effort as string)); +}); + +// The x-omniroute-effort header is read INSIDE the module from the incoming +// client request (chatCore.ts passes `clientRawRequest` through untouched), so +// the call site does not need its own header extraction. +test("header is read from clientRawRequest.headers (plain record, case-insensitive)", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + clientRawRequest: { headers: { "X-OmniRoute-Effort": "auto" } }, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out.reasoning_effort, "high"); +}); + +test("header is read from clientRawRequest.headers (Headers instance)", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: trivialMsgs }, + clientRawRequest: { headers: new Headers({ "x-omniroute-effort": "auto" }) }, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out.reasoning_effort, "low"); +}); + +test("clientRawRequest without the header (or without headers at all) is not an opt-in", () => { + const body = { model: "m" }; + const noHeader = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + clientRawRequest: { headers: { "user-agent": "x" } }, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(noHeader, body); + const noHeaders = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + clientRawRequest: {}, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(noHeaders, body); + const noRequest = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(noRequest, body); +}); + +test("a pre-extracted headerEffort takes precedence over clientRawRequest.headers", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: null, + clientRawRequest: { headers: { "x-omniroute-effort": "auto" } }, + targetFormat: FORMATS.OPENAI, + }); + assert.equal(out, body, "explicit null means the caller already decided: no opt-in"); +}); + +// #13448 rework: the field wireAdaptiveEffort injects (`reasoning_effort`) is an +// OpenAI Chat-Completions-shaped field. On any other dispatch format it is either +// inert (Claude/Gemini read `thinking`/`reasoning.effort` instead) or actively +// harmful (Anthropic's Messages API 400s on an unrecognized top-level parameter). +// Every sibling reasoning-shape normalization in chatCore.ts is scoped to +// `FORMATS.OPENAI` the same way (applyDefaultReasoningEffort, +// promoteStrayReasoningEffort's same-format Responses lane) -- wiring must match. +test("header opt-in is a no-op on a Claude-targeted dispatch (body returned unchanged)", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: "auto", + targetFormat: FORMATS.CLAUDE, + }); + assert.equal(out, body, "must be the exact same reference -- no reasoning_effort injected"); + assert.equal(out.reasoning_effort, undefined); +}); + +test("header opt-in is a no-op on a Gemini-targeted dispatch", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: "auto", + targetFormat: FORMATS.GEMINI, + }); + assert.equal(out, body); + assert.equal(out.reasoning_effort, undefined); +}); + +test("header read from clientRawRequest is also a no-op on a non-OpenAI target", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + clientRawRequest: { headers: { "x-omniroute-effort": "auto" } }, + targetFormat: FORMATS.CLAUDE, + }); + assert.equal(out, body); + assert.equal(out.reasoning_effort, undefined); +}); + +test("model-default 'auto' marker is left untouched (not stripped, not resolved) on a non-OpenAI target", () => { + // Guards against a partial fix that strips the "auto" marker before the + // targetFormat check -- on a non-OpenAI target the body (including any stray + // literal "auto") must be untouched, since it was never OmniRoute's own + // injection to interpret on that dispatch shape. + const body = { model: "m", reasoning_effort: "auto" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: null, + targetFormat: FORMATS.CLAUDE, + }); + assert.equal(out, body); + assert.equal(out.reasoning_effort, "auto"); +}); + +test("targetFormat undefined (e.g. an uncovered call site) also no-ops -- fail closed", () => { + const body = { model: "m" }; + const out = wireAdaptiveEffort(body, { + rawBody: { messages: heavyMsgs }, + headerEffort: "auto", + targetFormat: undefined, + }); + assert.equal(out, body); +}); diff --git a/tests/unit/adaptive-effort.test.ts b/tests/unit/adaptive-effort.test.ts new file mode 100644 index 00000000000..a86b9eb0d9b --- /dev/null +++ b/tests/unit/adaptive-effort.test.ts @@ -0,0 +1,89 @@ +// Adaptive effort tests — mirrors the Hermes contract (hermes-agent#109044): +// trivial→low, heavy→high, mid→medium; explicit client effort wins; the +// stateless per-turn pin ignores post-last-user tool traffic. +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + resolveAdaptiveEffort, + applyAdaptiveEffort, + isAdaptiveEffort, + hasExplicitReasoningField, +} from "@omniroute/open-sse/services/adaptiveEffort.ts"; + +function msg( + role: string, + content: string | unknown[] +): { role: string; content: string | unknown[] } { + return { role, content }; +} + +test("trivial ask resolves low", () => { + const level = resolveAdaptiveEffort([msg("user", "list the files")]); + assert.equal(level, "low"); +}); + +test("heavy context resolves high", () => { + const level = resolveAdaptiveEffort([ + msg("user", "long enough"), + msg("assistant", "x".repeat(40)), + msg("user", "continue " + "deep work ".repeat(400)), + ]); + assert.equal(level, "high"); +}); + +test("trivial ask inside big prior context is NOT low", () => { + const level = resolveAdaptiveEffort([ + msg("user", "first ask that is long enough to not be trivial itself " + "pad ".repeat(60)), + msg("assistant", "answer " + "y".repeat(30000)), + msg("user", "short follow-up"), + ]); + assert.equal(level, "medium"); +}); + +test("stateless pin: mid-tool-loop request resolves same as turn start", () => { + const userTurn = [msg("user", "fix the failing test")]; + const turnStart = resolveAdaptiveEffort(userTurn); + const midLoop = resolveAdaptiveEffort([ + ...userTurn, + msg("assistant", "checking"), + msg("tool", "z".repeat(5000)), + msg("assistant", "checking more"), + msg("tool", "z".repeat(5000)), + ]); + // Mid-loop: same turn → same level (pin), even though raw context grew. + assert.equal(midLoop, turnStart); +}); + +test("explicit effort wins over auto", () => { + const body = { messages: [msg("user", "hello there")], reasoning_effort: "high" }; + const out = applyAdaptiveEffort(body, { headerEffort: "auto", modelDefaultEffort: "auto" }); + assert.equal(out.reasoning_effort, "high"); + assert.equal(hasExplicitReasoningField(body), true); +}); + +test("header auto on trivial ask injects low", () => { + const body: Record = { messages: [msg("user", "hi")] }; + const out = applyAdaptiveEffort(body, { headerEffort: "auto" }); + assert.equal(out.reasoning_effort, "low"); +}); + +test("model default auto injects resolved level", () => { + const out = applyAdaptiveEffort({ messages: [msg("user", "hello")] } as Record, { + modelDefaultEffort: "auto", + }); + assert.equal(out.reasoning_effort, "low"); + assert.equal(isAdaptiveEffort("auto"), true); + assert.equal(isAdaptiveEffort("AUTO "), true); +}); + +test("no auto opt-in → unchanged reference", () => { + const body = { messages: [msg("user", "hello")] }; + const out = applyAdaptiveEffort(body, {}); + assert.equal(out, body); +}); + +test("empty messages resolves medium (safe default)", () => { + assert.equal(resolveAdaptiveEffort([]), "medium"); + assert.equal(resolveAdaptiveEffort(undefined), "medium"); + assert.equal(resolveAdaptiveEffort(null), "medium"); +}); diff --git a/tests/unit/devin-cloud-agent-validator-6142.test.ts b/tests/unit/devin-cloud-agent-validator-6142.test.ts index 39463beadbb..e20d7d1c6db 100644 --- a/tests/unit/devin-cloud-agent-validator-6142.test.ts +++ b/tests/unit/devin-cloud-agent-validator-6142.test.ts @@ -3,51 +3,83 @@ // generic Providers config page (parity with the existing jules cloud-agent wiring). import { test } from "node:test"; import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; import { validateProviderApiKey } from "../../src/lib/providers/validation.ts"; import { validateDevinCloudAgentProvider } from "../../src/lib/providers/validation/webProvidersB.ts"; import { getStaticModelsForProvider } from "../../src/lib/providers/staticModels.ts"; +// Writes a tiny executable shell script that inspects WINDSURF_API_KEY (the env +// var the real devin-cli executor uses, per open-sse/executors/devin-cli.ts) and +// exits 0 only when it matches `expectedKey`. Pointing CLI_DEVIN_BIN at this +// script exercises the REAL spawn() call in validateDevinCliKeyFallback — no +// module mocking — proving both that the fallback is invoked and that it wires +// the api key through the same env var the executor relies on. +function writeFakeDevinCli(expectedKey: string, exitCode: number): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-fake-devin-cli-")); + const scriptPath = path.join(dir, "devin"); + fs.writeFileSync( + scriptPath, + `#!/bin/sh\nif [ "$WINDSURF_API_KEY" = "${expectedKey}" ]; then\n exit ${exitCode}\nelse\n exit 1\nfi\n`, + { mode: 0o755 } + ); + return scriptPath; +} + test("#6142: devin cloud-agent validator is wired into the SPECIALTY_VALIDATORS dispatcher", async () => { const originalFetch = globalThis.fetch; - globalThis.fetch = (async () => - new Response("{}", { status: 401 })) as unknown as typeof fetch; + // #devin-cli-key: the 401 path falls back to probing the local Devin CLI. + // Point CLI_DEVIN_BIN at a nonexistent binary so the fallback fails + // deterministically regardless of the host machine's devin install. + const originalBin = process.env.CLI_DEVIN_BIN; + process.env.CLI_DEVIN_BIN = "/nonexistent/devin-for-tests"; + globalThis.fetch = (async () => new Response("{}", { status: 401 })) as unknown as typeof fetch; try { const result = await validateProviderApiKey({ provider: "devin", - apiKey: "cog_bad_key", + apiKey: "k", }); assert.equal(result.valid, false); assert.equal(result.error, "Invalid API key"); assert.notEqual(result.unsupported, true); } finally { globalThis.fetch = originalFetch; + if (originalBin === undefined) delete process.env.CLI_DEVIN_BIN; + else process.env.CLI_DEVIN_BIN = originalBin; } }); test("#6142: validateDevinCloudAgentProvider maps 401 to Invalid API key", async () => { const originalFetch = globalThis.fetch; - globalThis.fetch = (async () => - new Response("{}", { status: 401 })) as unknown as typeof fetch; + const originalBin = process.env.CLI_DEVIN_BIN; + process.env.CLI_DEVIN_BIN = "/nonexistent/devin-for-tests"; + globalThis.fetch = (async () => new Response("{}", { status: 401 })) as unknown as typeof fetch; try { - const result = await validateDevinCloudAgentProvider({ apiKey: "bad-key" }); + const result = await validateDevinCloudAgentProvider({ apiKey: "k" }); assert.equal(result.valid, false); assert.equal(result.error, "Invalid API key"); } finally { globalThis.fetch = originalFetch; + if (originalBin === undefined) delete process.env.CLI_DEVIN_BIN; + else process.env.CLI_DEVIN_BIN = originalBin; } }); test("#6142: validateDevinCloudAgentProvider maps 403 to Invalid API key", async () => { const originalFetch = globalThis.fetch; - globalThis.fetch = (async () => - new Response("{}", { status: 403 })) as unknown as typeof fetch; + const originalBin = process.env.CLI_DEVIN_BIN; + process.env.CLI_DEVIN_BIN = "/nonexistent/devin-for-tests"; + globalThis.fetch = (async () => new Response("{}", { status: 403 })) as unknown as typeof fetch; try { - const result = await validateDevinCloudAgentProvider({ apiKey: "bad-key" }); + const result = await validateDevinCloudAgentProvider({ apiKey: "k" }); assert.equal(result.valid, false); assert.equal(result.error, "Invalid API key"); } finally { globalThis.fetch = originalFetch; + if (originalBin === undefined) delete process.env.CLI_DEVIN_BIN; + else process.env.CLI_DEVIN_BIN = originalBin; } }); @@ -64,6 +96,44 @@ test("#6142: validateDevinCloudAgentProvider accepts a 2xx probe as valid", asyn } }); +test("#devin-cli-key: a CLI-format key rejected by the HTTP API validates via the real CLI probe", async () => { + const originalFetch = globalThis.fetch; + const originalBin = process.env.CLI_DEVIN_BIN; + const apiKey = "apk_user_cli_format_key"; + const fakeCliPath = writeFakeDevinCli(apiKey, 0); + process.env.CLI_DEVIN_BIN = fakeCliPath; + globalThis.fetch = (async () => new Response("{}", { status: 401 })) as unknown as typeof fetch; + try { + const result = await validateDevinCloudAgentProvider({ apiKey }); + assert.equal(result.valid, true); + assert.equal(result.error, null); + assert.match(result.warning ?? "", /Devin CLI/); + } finally { + globalThis.fetch = originalFetch; + if (originalBin === undefined) delete process.env.CLI_DEVIN_BIN; + else process.env.CLI_DEVIN_BIN = originalBin; + fs.rmSync(path.dirname(fakeCliPath), { recursive: true, force: true }); + } +}); + +test("#devin-cli-key: the CLI fallback still rejects when the CLI probe itself fails (wrong key)", async () => { + const originalFetch = globalThis.fetch; + const originalBin = process.env.CLI_DEVIN_BIN; + const fakeCliPath = writeFakeDevinCli("the-correct-key", 0); + process.env.CLI_DEVIN_BIN = fakeCliPath; + globalThis.fetch = (async () => new Response("{}", { status: 401 })) as unknown as typeof fetch; + try { + const result = await validateDevinCloudAgentProvider({ apiKey: "a-different-key" }); + assert.equal(result.valid, false); + assert.equal(result.error, "Invalid API key"); + } finally { + globalThis.fetch = originalFetch; + if (originalBin === undefined) delete process.env.CLI_DEVIN_BIN; + else process.env.CLI_DEVIN_BIN = originalBin; + fs.rmSync(path.dirname(fakeCliPath), { recursive: true, force: true }); + } +}); + test("#6142: devin exposes a static model catalog for the 'Available Models' UI (parity with jules)", () => { const models = getStaticModelsForProvider("devin"); assert.ok(Array.isArray(models) && models.length > 0); diff --git a/tests/unit/vacuum-reclaimable-threshold.test.ts b/tests/unit/vacuum-reclaimable-threshold.test.ts new file mode 100644 index 00000000000..0dbd7707f97 --- /dev/null +++ b/tests/unit/vacuum-reclaimable-threshold.test.ts @@ -0,0 +1,132 @@ +// The auto-cleanup scheduler used to run a full `VACUUM` after ANY deletion +// (`totalDeleted > 0`), no matter how small. `db.exec("VACUUM")` is synchronous +// and blocks the entire process -- on a multi-GB database that freezes all HTTP +// traffic for 15-20 minutes. Observed live: a routine cleanup that freed 2-6 +// rows re-triggered a full VACUUM on every restart, because the new terminal- +// batch cleanup (see db-terminal-batch-and-file-cleanup.test.ts) almost always +// finds a handful of newly-aged-out batches. +// +// Row count was never the right signal anyway: a few oversized +// batch_item_checkpoints rows can free far more space than thousands of tiny +// audit-log rows. This pins vacuumAfterCleanup()'s reclaimable-bytes gate: SQLite's +// own free-page count (PRAGMA freelist_count), not "were any rows deleted". The +// row-count gate (OMNIROUTE_VACUUM_MIN_DELETED_ROWS) is untouched and still fires +// on its own -- see db-cleanup-vacuum-gate.test.ts -- this file only pins the +// ADDITIONAL reclaimable-bytes trigger. + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-vacuum-threshold-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const cleanup = await import("../../src/lib/db/cleanup.ts"); + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +async function withEnv( + key: string, + value: string | undefined, + fn: () => T | Promise +): Promise { + const prev = process.env[key]; + if (value === undefined) delete process.env[key]; + else process.env[key] = value; + try { + return await fn(); + } finally { + if (prev === undefined) delete process.env[key]; + else process.env[key] = prev; + } +} + +test("getVacuumMinReclaimableBytes: defaults to 100 MB", async () => { + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", undefined, () => { + assert.equal(cleanup.getVacuumMinReclaimableBytes(), 100 * 1024 * 1024); + }); +}); + +test("getVacuumMinReclaimableBytes: honors the env override, including 0 (always vacuum)", async () => { + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", "5", () => { + assert.equal(cleanup.getVacuumMinReclaimableBytes(), 5 * 1024 * 1024); + }); + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", "0", () => { + assert.equal(cleanup.getVacuumMinReclaimableBytes(), 0); + }); +}); + +test("getVacuumMinReclaimableBytes: rejects garbage and falls back to the default", async () => { + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", "not-a-number", () => { + assert.equal(cleanup.getVacuumMinReclaimableBytes(), 100 * 1024 * 1024); + }); + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", "-1", () => { + assert.equal(cleanup.getVacuumMinReclaimableBytes(), 100 * 1024 * 1024); + }); +}); + +test("vacuumAfterCleanup: reclaimable-bytes gate skips VACUUM when space is under the threshold (row gate also below threshold)", async () => { + const db = core.getDbInstance(); + db.exec("CREATE TABLE IF NOT EXISTS vacuum_threshold_probe (id INTEGER PRIMARY KEY, v TEXT)"); + // A handful of tiny rows leaves negligible freelist space after deletion -- + // nowhere near the (very high, deliberately unreachable in this test) threshold. + db.exec("INSERT INTO vacuum_threshold_probe (v) VALUES ('a'), ('b'), ('c')"); + db.exec("DELETE FROM vacuum_threshold_probe"); + + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", "999999", async () => { + // Also keep the row-count gate from firing on its own so only the + // reclaimable-bytes gate is under test here. + await withEnv("OMNIROUTE_VACUUM_MIN_DELETED_ROWS", "999999", async () => { + // Must not throw, and specifically must not attempt to run VACUUM at all -- + // proven by the fact that a VACUUM would otherwise reset freelist_count. + const before = cleanup.getReclaimableBytes(db); + const ran = await cleanup.vacuumAfterCleanup( + 3, + (sql) => db.exec(sql), + () => {}, + () => {}, + () => cleanup.getReclaimableBytes(db) + ); + const after = cleanup.getReclaimableBytes(db); + assert.equal(ran, false); + assert.equal(after, before, "skipping VACUUM must leave the freelist untouched"); + }); + }); +}); + +test("vacuumAfterCleanup: reclaimable-bytes gate alone triggers VACUUM even when the row-count gate is not met", async () => { + const db = core.getDbInstance(); + db.exec("CREATE TABLE IF NOT EXISTS vacuum_threshold_probe2 (id INTEGER PRIMARY KEY, v TEXT)"); + const insert = db.prepare("INSERT INTO vacuum_threshold_probe2 (v) VALUES (?)"); + const big = "x".repeat(4096); + for (let i = 0; i < 200; i++) insert.run(big); + db.exec("DELETE FROM vacuum_threshold_probe2"); + + const reclaimableBeforeVacuum = cleanup.getReclaimableBytes(db); + assert.ok(reclaimableBeforeVacuum > 0, "sanity: the delete above must have freed some pages"); + + await withEnv("OMNIROUTE_VACUUM_MIN_RECLAIMABLE_MB", "0", async () => { + // Row-count gate deliberately unreachable: only 1 row "deleted" here, far + // below any realistic OMNIROUTE_VACUUM_MIN_DELETED_ROWS value, proving the + // reclaimable-bytes signal alone is sufficient to trigger VACUUM. + await withEnv("OMNIROUTE_VACUUM_MIN_DELETED_ROWS", "999999", async () => { + const ran = await cleanup.vacuumAfterCleanup( + 1, + (sql) => db.exec(sql), + () => {}, + () => {}, + () => cleanup.getReclaimableBytes(db) + ); + assert.equal(ran, true); + }); + }); + + // A successful VACUUM rebuilds the file with no free pages left over. + assert.equal(cleanup.getReclaimableBytes(db), 0, "VACUUM must have actually run"); +});