From 01ab5d1fd5e2986f6c482647136a69a4ed7b5c36 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Tue, 14 Jul 2026 13:28:51 -0300 Subject: [PATCH 01/14] =?UTF-8?q?chore(ci):=20add=20.mergify.yml=20to=20ma?= =?UTF-8?q?in=20=E2=80=94=20Mergify=20only=20reads=20config=20from=20the?= =?UTF-8?q?=20default=20branch=20(#7168)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .mergify.yml | 55 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) create mode 100644 .mergify.yml diff --git a/.mergify.yml b/.mergify.yml new file mode 100644 index 00000000000..131c6d71a99 --- /dev/null +++ b/.mergify.yml @@ -0,0 +1,55 @@ +# Mergify merge queue — WS3.4/D5 of the v3.8.49 quality/velocity master plan. +# +# WHY: ~85-100 active PR authors/month and 300+ PRs/week peaks, all merged by ONE +# identity. The manual merge-train validated batches by hand; this queue automates +# it with batching + automatic batch bisection (a red batch of N costs ~log2(N) +# revalidations instead of N). Mergify Open Source plan: free, unlimited, public repo. +# +# GOVERNANCE (non-negotiable, mirrors CLAUDE.md Hard Rules #21/#22 + the owner's +# pre-merge ⭐ gate): +# • A PR enters the queue ONLY via the `queue` label — applied by the owner (or a +# session acting for the owner) AFTER the pre-merge ⭐ report/decision. The label +# IS the merge approval; Mergify only executes it. +# • During a release-freeze (open issue labeled `release-freeze`), do NOT label PRs +# targeting the frozen branch — the freeze is a human-honored coordination signal +# the queue cannot see. Retarget to the active release/vX+1 first (Hard Rule #21). +# • Never label a PR another session is actively working (Hard Rule #22b). +# • Fallback path if Mergify misbehaves or the OSS plan changes: the manual +# merge-train runbook (docs/ops/MERGE_TRAIN.md) — remove labels, proceed by hand. + +queue_rules: + - name: release + # Any current or future release branch — the reason GitHub's native queue was + # rejected (no wildcard support on personal-account repos). + queue_conditions: + - base~=^release/v\d+\.\d+\.\d+$ + - label=queue + - -draft + - -conflict + # "Everything that ran is green, nothing still running, AND the always-on + # anchor check succeeded" — robust to the path-filtered fast-gates (docs-only + # PRs skip code jobs; matrix shard names vary) while never fail-open: a PR with + # zero checks cannot vacuously merge, because `Merge integrity` runs on EVERY + # non-draft PR (quality.yml) and must be an affirmative success. Review approval + # is intentionally NOT a condition here: the owner-applied `queue` label IS the + # approval in this repo's single-maintainer model (see governance header). + merge_conditions: + - "#check-failure=0" + - "#check-pending=0" + - "#check-success>=1" + - check-success=Merge integrity (changelog + generated skills) + # Batching: validate up to 10 queued PRs together (the manual train's sweet spot); + # don't hold a lone PR hostage waiting for siblings. + batch_size: 10 + batch_max_wait_time: 5 min + # Squash keeps the one-commit-per-PR history the CHANGELOG reconciliation expects. + merge_method: squash + +pull_request_rules: + - name: clean up the queue label after merge + conditions: + - merged + actions: + label: + remove: + - queue From fb24740b73443c33e5e6f7ee8b781fa92e5e4cc1 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Tue, 14 Jul 2026 19:11:35 -0300 Subject: [PATCH 02/14] fix(ci): add the auto-enqueue pull_request_rule to the Mergify config (queue_conditions alone are eligibility-only) (#7179) --- .mergify.yml | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/.mergify.yml b/.mergify.yml index 131c6d71a99..88bc431369e 100644 --- a/.mergify.yml +++ b/.mergify.yml @@ -46,6 +46,18 @@ queue_rules: merge_method: squash pull_request_rules: + # AUTO-ENQUEUE trigger — queue_conditions alone only define ELIGIBILITY in current + # Mergify semantics (without this rule the PR sits at "use @Mergifyio queue", + # observed live on PR #7175). The conditions mirror queue_conditions on purpose. + - name: queue on owner-applied label + conditions: + - base~=^release/v\d+\.\d+\.\d+$ + - label=queue + - -draft + - -conflict + actions: + queue: + name: release - name: clean up the queue label after merge conditions: - merged From fb25ebdeb0dbd924d6c7f95fd284759e3f87d7c2 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Tue, 14 Jul 2026 23:49:49 -0300 Subject: [PATCH 03/14] fix(ci): migrate Mergify auto-enqueue to merge_protections_settings.auto_merge_conditions (rules-based path is EOL 2026-07-16) (#7216) --- .mergify.yml | 19 +++++++------------ 1 file changed, 7 insertions(+), 12 deletions(-) diff --git a/.mergify.yml b/.mergify.yml index 88bc431369e..d7eb728c0fd 100644 --- a/.mergify.yml +++ b/.mergify.yml @@ -17,6 +17,13 @@ # • Fallback path if Mergify misbehaves or the OSS plan changes: the manual # merge-train runbook (docs/ops/MERGE_TRAIN.md) — remove labels, proceed by hand. +# Auto-enqueue (current Mergify model, 2026): auto_merge_conditions in +# merge_protections_settings — the rules-based queue action / autoqueue path is +# deprecated (EOL 2026-07-16). The owner-applied `queue` label IS the approval. +merge_protections_settings: + auto_merge_conditions: + - label = queue + queue_rules: - name: release # Any current or future release branch — the reason GitHub's native queue was @@ -46,18 +53,6 @@ queue_rules: merge_method: squash pull_request_rules: - # AUTO-ENQUEUE trigger — queue_conditions alone only define ELIGIBILITY in current - # Mergify semantics (without this rule the PR sits at "use @Mergifyio queue", - # observed live on PR #7175). The conditions mirror queue_conditions on purpose. - - name: queue on owner-applied label - conditions: - - base~=^release/v\d+\.\d+\.\d+$ - - label=queue - - -draft - - -conflict - actions: - queue: - name: release - name: clean up the queue label after merge conditions: - merged From a7956945355d8b4dbdf9df531f5f53fb4d4d4274 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Wed, 15 Jul 2026 00:16:02 -0300 Subject: [PATCH 04/14] fix(ci): drop Mergify batch settings (batching is a paid-tier feature; free plan queue is serial) (#7220) --- .mergify.yml | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/.mergify.yml b/.mergify.yml index d7eb728c0fd..f7cbe19d7ea 100644 --- a/.mergify.yml +++ b/.mergify.yml @@ -45,10 +45,9 @@ queue_rules: - "#check-pending=0" - "#check-success>=1" - check-success=Merge integrity (changelog + generated skills) - # Batching: validate up to 10 queued PRs together (the manual train's sweet spot); - # don't hold a lone PR hostage waiting for siblings. - batch_size: 10 - batch_max_wait_time: 5 min + # NO batching: 'Merge Queue Batch' requires a paid Mergify tier (live finding + # 2026-07-15 — the queue command fails with "Cannot use Merge Queue batch" on + # the free plan). Serial queue (1 PR at a time) still automates the train. # Squash keeps the one-commit-per-PR history the CHANGELOG reconciliation expects. merge_method: squash From 9875ccf4e62fb5a26e54f712e192647293bf646b Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Wed, 15 Jul 2026 00:48:07 -0300 Subject: [PATCH 05/14] fix(ci): merge queue tolerates the advisory dast-smoke failure (its GH-hosted build hang dequeued every attempt) (#7225) --- .mergify.yml | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/.mergify.yml b/.mergify.yml index f7cbe19d7ea..7cc4882127e 100644 --- a/.mergify.yml +++ b/.mergify.yml @@ -41,7 +41,14 @@ queue_rules: # is intentionally NOT a condition here: the owner-applied `queue` label IS the # approval in this repo's single-maintainer model (see governance header). merge_conditions: - - "#check-failure=0" + # "Zero failures" — EXCEPT the advisory dast-smoke: it is continue-on-error by + # design and its GH-hosted CLI build hangs recurrently (drafts #7184/#7221 were + # dequeued solely by it). Any OTHER failure still blocks (anti-fail-open kept). + - or: + - "#check-failure=0" + - and: + - "#check-failure=1" + - check-failure=dast-smoke - "#check-pending=0" - "#check-success>=1" - check-success=Merge integrity (changelog + generated skills) From 0065f1b7f03d0cfa27a42491e92526e73d9b44fd Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Thu, 16 Jul 2026 00:12:29 -0300 Subject: [PATCH 06/14] =?UTF-8?q?test(ci):=20make=20the=20#6634=20selfref?= =?UTF-8?q?=20guard=20hermetic=20=E2=80=94=20main's=20copy=20hard-fails=20?= =?UTF-8?q?every=20PR=20(#7341)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit main's copy of this test still does git I/O inside a unit test: const baseSrc = git(['show', 'origin/main:' + FILE]); Runners check out a shallow single ref, so origin/main does not resolve and the test dies with 'fatal: invalid object name origin/main'. Every PR into main fails Unit Tests (7/8) on it — today that is #7313, #7315, #7316, #7334, #7336 and #7337, six PRs red on a defect none of them introduced. #7313 has no other red at all. release/v3.8.49 already carries a fix (2e42b8efc, #7174: try/catch, fetch origin/main on demand, t.skip() when unreachable), but it only reaches main at release time — so main stays broken for the whole cycle. Cherry-picking it would also import a new problem: PR Test Policy classifies t.skip() as a silenced assertion, which we watched it correctly catch on #7300 today. This is the hermetic version instead (ported from #7327, which does the same for the release branch): read the file straight off disk, compare against an empty base so baseTaut/baseExtTaut are 0 — the strictest possible comparison point — and call evaluateMasking() directly. No git ref, no fetch, no skip, nothing the runner's checkout depth can break. The #6634 regression stays covered: the guard's logic lives in SELF_TEST_FIXTURE_RE (check-test-masking.mjs:337), not in the test. Proven both ways on main before committing — neutralise SELF_TEST_FIXTURE_RE to /$^/ and the test FAILS; restore it and it passes 2/2, with check-test-masking.mjs left byte-identical. Co-authored-by: growab --- .../check-test-masking-selfref-6634.test.ts | 23 +++++++++++-------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/tests/unit/check-test-masking-selfref-6634.test.ts b/tests/unit/check-test-masking-selfref-6634.test.ts index 6d91a2e7c21..b233a008f22 100644 --- a/tests/unit/check-test-masking-selfref-6634.test.ts +++ b/tests/unit/check-test-masking-selfref-6634.test.ts @@ -14,12 +14,14 @@ * (`if (file.endsWith("check-test-masking.test.ts")) continue;` in * scripts/check/check-test-masking.mjs) for precisely this reason — this test * asserts evaluateMasking() now applies the same exclusion for its diff-based - * tautology counters, using the real base(origin/main)/head(HEAD) diff of + * tautology counters, against the REAL current source of * tests/unit/check-test-masking.test.ts. */ import test from "node:test"; import assert from "node:assert/strict"; -import { execFileSync } from "node:child_process"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; import { countTautologies, @@ -28,16 +30,17 @@ import { } from "../../scripts/check/check-test-masking.mjs"; const FILE = "tests/unit/check-test-masking.test.ts"; - -function git(args: string[]): string { - return execFileSync("git", args, { encoding: "utf8" }); -} +const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); test("#6634: check-test-masking.test.ts's own tautology fixtures must not self-flag as weakening", () => { - // origin/main predates the #6404 fixtures (countBareTautologies/scanBareTautologies - // tests) that legitimately embed tautology-pattern literals as string fixtures. - const baseSrc = git(["show", "origin/main:" + FILE]); - const headSrc = git(["show", "HEAD:" + FILE]); + // Read the REAL current source from disk rather than a git ref: the Unit Tests + // job checks out a shallow/single-ref tree with no origin/main, so `git show + // origin/main:` failed the shard before it ever exercised the masking + // behavior under test. An empty base models the file's pre-#6404 state (no + // fixtures), which maximizes headTaut - baseTaut — the strictest input for the + // exclusion this test asserts. + const baseSrc = ""; + const headSrc = fs.readFileSync(path.join(REPO_ROOT, FILE), "utf8"); const perFile = [ { From ddd6d09cd10d7a2d6af793a0bfe20d96c05db25c Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Thu, 16 Jul 2026 00:47:07 -0300 Subject: [PATCH 07/14] chore(quality): tighten main's coverage baseline to the CI's real numbers (#7347) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit main's ratchet had been failing --require-tighten on every PR: 11 metrics improved but the baseline was never tightened. Same class as the #6634 selfref guard — an infra fix that lands only on the release branch leaves main red for the whole cycle, and every PR into main pays for it. Values are the merged-coverage numbers from a run on main itself (a local run measures ~68% vs CI's ~80%; the baseline's own note warns about that gap). Only the 11 coverage values change — gitleaks and semgrepFindings keep main's own state. No changelog fragment: #7326 carries it on release/v3.8.49, and a second one here would double the entry at release time. --- config/quality/quality-baseline.json | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index f660fc33d30..943b0fa4b88 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -27,22 +27,22 @@ "eps": 0 }, "coverage.statements": { - "value": 76.5, + "value": 80.8, "direction": "up", "tightenSlack": 5 }, "coverage.lines": { - "value": 76.5, + "value": 80.8, "direction": "up", "tightenSlack": 5 }, "coverage.functions": { - "value": 82, + "value": 86.44, "direction": "up", "tightenSlack": 5 }, "coverage.branches": { - "value": 73, + "value": 78.1, "direction": "up", "eps": 1.5, "tightenSlack": 5 @@ -54,43 +54,43 @@ "tightenSlack": 10 }, "coverage.combo.lines": { - "value": 80, + "value": 85.42, "direction": "up", "eps": 1.5, "tightenSlack": 10 }, "coverage.accountFallback.lines": { - "value": 88, + "value": 96.78, "direction": "up", "eps": 1.5, "tightenSlack": 10 }, "coverage.auth.lines": { - "value": 90, + "value": 92.55, "direction": "up", "eps": 1.5, "tightenSlack": 10 }, "coverage.routeGuard.lines": { - "value": 94, + "value": 98.73, "direction": "up", "eps": 1.5, "tightenSlack": 10 }, "coverage.error.lines": { - "value": 88, + "value": 92.13, "direction": "up", "eps": 1.5, "tightenSlack": 10 }, "coverage.publicCreds.lines": { - "value": 92, + "value": 99.07, "direction": "up", "eps": 1.5, "tightenSlack": 10 }, "coverage.circuitBreaker.lines": { - "value": 92, + "value": 95.09, "direction": "up", "eps": 1.5, "tightenSlack": 10 From 37cad3eea50f7860e282b54cff8d8f030b45b75e Mon Sep 17 00:00:00 2001 From: Minxi Hou Date: Thu, 16 Jul 2026 00:48:24 -0400 Subject: [PATCH 08/14] fix(antigravity): remove hardcoded 120s SSE collect timeout The SSE collection in collectStreamToResponse had a hardcoded 120 s timeout. Reasoning-heavy models like gemini-3.1-pro-high on large prompts (>30 KB) regularly exceed 120 s of generation time, causing the executor to return a synthetic 504 before the model finishes. Replace the hardcoded value with FETCH_TIMEOUT_MS (default 600 s, overridable via FETCH_TIMEOUT_MS env var), which is the standard upstream-request budget across all OmniRoute providers. Signed-off-by: Minxi Hou --- open-sse/executors/antigravity.ts | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index 494d750836e..83b014002dc 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -13,6 +13,7 @@ import { PROVIDERS, OAUTH_ENDPOINTS, HTTP_STATUS, + FETCH_TIMEOUT_MS, STREAM_READINESS_TIMEOUT_MS, ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE, } from "../config/constants.ts"; @@ -948,7 +949,11 @@ export class AntigravityExecutor extends BaseExecutor { const decoder = new TextDecoder(); const logger = log || undefined; - const SSE_COLLECT_TIMEOUT_MS = 120_000; + // Guard against indefinite hangs when the upstream sends headers but + // stalls on the body. Inherit the global FETCH_TIMEOUT_MS (default 600 s, + // overridable via env) so reasoning-heavy models (gemini-3.1-pro-high on + // large prompts) are not killed by a hardcoded 120 s ceiling. + const SSE_COLLECT_TIMEOUT_MS = FETCH_TIMEOUT_MS; const collect = async () => { const collected: AntigravityCollectedStream = { From e33937eef3583eeab8c8e417ae871ca5546a158f Mon Sep 17 00:00:00 2001 From: Minxi Hou Date: Sat, 18 Jul 2026 07:16:17 -0400 Subject: [PATCH 09/14] fix(antigravity): streaming passthrough for non-streaming clients When a client sends stream: false to the Antigravity executor (Gemini models), OmniRoute buffered the entire SSE stream before responding. Long-thinking models exceeded the 120s timeout. Remove hardcoded SSE_COLLECT_TIMEOUT_MS. Extract shared createCreditsExtractionTransform with 16KB buffer cap and abort handling for client disconnect. Add parseSSEToGeminiResponse for the non-streaming drain path. Fix hasGeminiTerminalFinishReason to check top-level candidates (no response wrapper). Add signal null guards for credits retry path. Return 499 on early abort instead of piping cancelled body. Also remove duplicate SKILLS_SANDBOX_RUNTIME from .env.example and clarify .artifacts/ vs _artifacts/ in .gitignore. Signed-off-by: Minxi Hou --- .env.example | 2 +- .gitignore | 4 +- config/quality/eslint-suppressions.json | 5 - open-sse/executors/antigravity.ts | 302 ++++++++++-------- open-sse/handlers/chatCore/nonStreamingSse.ts | 43 ++- open-sse/handlers/sseParser.ts | 153 +++++++++ tests/unit/executor-antigravity.test.ts | 120 ++++++- tests/unit/sse-parser.test.ts | 129 +++++++- 8 files changed, 599 insertions(+), 159 deletions(-) diff --git a/.env.example b/.env.example index 60b5074e6e7..bfc6942f2dc 100644 --- a/.env.example +++ b/.env.example @@ -222,7 +222,7 @@ CONTAINER_HOST=docker # - orbstack: OrbStack (high-perf Linux VM + docker shim on macOS) # - podman: Podman (rootless, daemonless) # - docker: Docker (default fallback) -SKILLS_SANDBOX_RUNTIME=auto +# (defined under SKILLS & SANDBOXING section below) # ═══════════════════════════════════════════════════════════════════════════════ # 4. SECURITY & AUTHENTICATION diff --git a/.gitignore b/.gitignore index 67d69b3d5d6..e3064a2700e 100644 --- a/.gitignore +++ b/.gitignore @@ -232,7 +232,7 @@ omniroute.md # mise configuration mise.toml -_artifacts/ +_artifacts/ # release-green artifacts .claude-flow/ # ESLint file cache (npm run lint --cache / complexity ratchets) @@ -240,5 +240,5 @@ _artifacts/ .eslintcache-complexity -# CI/local quality artifacts (eslint-results.json, etc.) +# CI/local quality artifacts (eslint-results.json, quality-ratchet.md, etc.) .artifacts/ diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index dcea92f2a98..5cf9b231b26 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -2057,11 +2057,6 @@ "count": 13 } }, - "tests/unit/sse-parser.test.ts": { - "@typescript-eslint/no-explicit-any": { - "count": 15 - } - }, "tests/unit/startup-stale-cooldown-recovery.test.ts": { "@typescript-eslint/no-explicit-any": { "count": 1 diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index 83b014002dc..3647e82df07 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -272,6 +272,91 @@ export function updateAntigravityRemainingCredits(accountId: string, balance: nu } catch {} } +/** + * Create a pass-through TransformStream that extracts `remainingCredits` + * from SSE data without consuming the stream. The downstream client + * receives the unmodified bytes. + * + * @param accountId Provider account ID for credit-balance persistence. + * @param bufferSize Optional sliding-window buffer cap in bytes. + * Pass 0 or omit for unlimited (non-streaming callers + * where the full body is already buffered upstream). + * The streaming path uses 16384 (16 KB) to prevent OOM + * on long-lived SSE connections. Credit-balance data + * appears near the end of the SSE stream (after + * content), so the sliding window captures it even at + * 16 KB -- only truly massive responses (>16 KB of + * consecutive non-newline content) would lose credits. + * @internal Exported for unit testing only. + */ +export function createCreditsExtractionTransform( + accountId: string, + bufferSize = 0 +): TransformStream { + let buffer = ""; + const decoder = new TextDecoder(); + + return new TransformStream( + { + transform(chunk, controller) { + controller.enqueue(chunk); + try { + buffer += decoder.decode(chunk, { stream: true }); + // Sliding-window cap: truncate after the last complete newline + // in the discard region so SSE lines are never split mid-payload. + if (bufferSize > 0 && buffer.length > bufferSize) { + const lastNewline = buffer.lastIndexOf("\n", buffer.length - bufferSize); + if (lastNewline !== -1) { + buffer = buffer.slice(lastNewline + 1); + } else { + // No newline in the discard region -- incomplete line, discard entirely. + buffer = ""; + } + } + } catch { + /* decoding best-effort */ + } + }, + flush() { + try { + buffer += decoder.decode(); + } catch { + /* decoding best-effort */ + } + try { + const lines = buffer.split("\n"); + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + try { + const parsed = JSON.parse(payload); + if (Array.isArray(parsed?.remainingCredits)) { + const googleCredit = parsed.remainingCredits.find((c: unknown) => { + const credit = asRecord(c); + return credit?.creditType === "GOOGLE_ONE_AI"; + }) as AntigravityCreditEntry | undefined; + if (googleCredit) { + const balance = parseInt(String(googleCredit.creditAmount ?? ""), 10); + if (!isNaN(balance)) updateAntigravityRemainingCredits(accountId, balance); + } + } + } catch { + /* skip malformed lines */ + } + } + } catch { + /* credits extraction is best-effort */ + } + buffer = ""; + }, + }, + { highWaterMark: 16384 }, + { highWaterMark: 16384 } + ); +} + function isCreditsExhausted(accountId: string): boolean { const until = creditsExhaustedUntil.get(accountId); if (!until) return false; @@ -931,6 +1016,10 @@ export class AntigravityExecutor extends BaseExecutor { * Collect an SSE streaming response into a single non-streaming JSON response. * Parses Gemini-format SSE chunks and assembles text content + usage into one * OpenAI-format chat.completion payload. + * + * @deprecated Use the non-streaming SSE path in chatCore instead, which calls + * parseSSEToGeminiResponse() from sseParser.ts. This method is retained only + * for backward compatibility and may be removed in a future release. */ collectStreamToResponse( response: Response, @@ -1375,33 +1464,40 @@ export class AntigravityExecutor extends BaseExecutor { }); if (creditsResp.ok || creditsResp.status !== HTTP_STATUS.RATE_LIMITED) { log?.info?.("AG_CREDITS", `Credits retry succeeded: ${creditsResp.status}`); - if (!stream) { - const collected = await this.collectStreamToResponse( - creditsResp, - model, - url, - finalCreditsHeaders, - creditsBody, - log, - signal + if (!stream && creditsResp.body) { + // Client already disconnected — skip pipe + if (signal?.aborted) { + creditsResp.body.cancel().catch(() => {}); + return { + response: new Response(null, { status: 499 }), + url, + headers: finalCreditsHeaders, + transformedBody: null, + }; + } + // Return raw SSE with credits extraction + const crTransform = createCreditsExtractionTransform( + accountId, + 16 * 1024 // 16KB cap ); - // Parse _remainingCredits from the synthetic response and cache - try { - const syntheticJson = await collected.response.clone().json(); - const rc = syntheticJson?._remainingCredits; - if (Array.isArray(rc)) { - const googleCredit = rc.find((c) => c.creditType === "GOOGLE_ONE_AI"); - if (googleCredit) { - const balance = parseInt(googleCredit.creditAmount, 10); - if (!isNaN(balance)) - updateAntigravityRemainingCredits(accountId, balance); - } - } - } catch { - /**/ + const crTappedBody = creditsResp.body.pipeThrough(crTransform); + if (signal) { + signal.addEventListener( + "abort", + () => { + creditsResp.body.cancel().catch(() => {}); + }, + { once: true } + ); } return { - ...collected, + response: new Response(crTappedBody, { + status: creditsResp.status, + statusText: creditsResp.statusText, + headers: creditsResp.headers, + }), + url, + headers: finalCreditsHeaders, transformedBody: attachToolNameMap(creditsBody, requestToolNameMap), }; } @@ -1537,12 +1633,16 @@ export class AntigravityExecutor extends BaseExecutor { } } - // For non-streaming clients, collect the SSE stream and return a synthetic - // non-streaming Response so chatCore doesn't need to handle SSE conversion. + // For non-streaming clients, return the raw SSE stream with a + // credits-extraction TransformStream. chatCore's non-streaming path + // (readNonStreamingResponseBody + parseNonStreamingSSEPayload with + // Gemini format support) handles draining and conversion to JSON. + // This replaces the previous collectStreamToResponse() approach which + // had an artificial timeout (now the standard FETCH_BODY_TIMEOUT_MS + // of 10 min applies). if (!stream) { // #3229: surface a real upstream error instead of masking a 4xx/5xx as an - // empty `chat.completion` envelope (collectStreamToResponse synthesizes a - // success-shaped body when the upstream returned no SSE data). + // empty `chat.completion` envelope. if (!response.ok) { const rawBody = await response .clone() @@ -1563,35 +1663,53 @@ export class AntigravityExecutor extends BaseExecutor { transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), }; } - const collected = await this.collectStreamToResponse( - response, - model, - url, - finalHeaders, - transformedBody, - log, - signal - ); - // When credits were injected (credits-first or credits-retry), the - // synthetic body contains _remainingCredits — mirror it into the - // balance cache so the dashboard stays fresh. - try { - const syntheticJson = await collected.response.clone().json(); - const rc = syntheticJson?._remainingCredits; - if (Array.isArray(rc)) { - const googleCredit = rc.find( - (c: { creditType?: string }) => c?.creditType === "GOOGLE_ONE_AI" + + if (response.body) { + // Client already disconnected — skip pipe + if (signal?.aborted) { + response.body.cancel().catch(() => {}); + return { + response: new Response(null, { status: 499 }), + url, + headers: finalHeaders, + transformedBody: null, + }; + } + // Cancel upstream body on client disconnect + if (signal) { + signal.addEventListener( + "abort", + () => { + response.body?.cancel().catch(() => {}); + }, + { once: true } ); - if (googleCredit) { - const balance = parseInt(googleCredit.creditAmount, 10); - if (!isNaN(balance)) updateAntigravityRemainingCredits(accountId, balance); - } } - } catch { - /* balance cache is best-effort */ + + // Tap the stream to extract remainingCredits while passing + // data through unmodified. chatCore drains the full body. + const nsPassThrough = createCreditsExtractionTransform( + accountId, + 16 * 1024 // 16KB sliding-window cap to prevent OOM + ); + const tappedBody = response.body.pipeThrough(nsPassThrough); + return { + response: new Response(tappedBody, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }), + url, + headers: finalHeaders, + transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), + }; } + + // No body -- return as-is return { - ...collected, + response, + url, + headers: finalHeaders, transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), }; } @@ -1615,81 +1733,9 @@ export class AntigravityExecutor extends BaseExecutor { } } - let sseBuffer = ""; - const decoder = new TextDecoder(); // Singleton for correct streaming decode - const MAX_BUFFER_SIZE = 16 * 1024; // Limit to prevent OOM on large streams - - const passThrough = new TransformStream( - { - transform(chunk, controller) { - controller.enqueue(chunk); - // Accumulate text to scan for remainingCredits - try { - const text = decoder.decode(chunk, { stream: true }); - sseBuffer += text; - // Limit buffer size to prevent unbounded growth - // Truncate only after a complete newline to avoid splitting SSE lines mid-payload - if (sseBuffer.length > MAX_BUFFER_SIZE) { - const lastNewline = sseBuffer.lastIndexOf( - "\n", - sseBuffer.length - MAX_BUFFER_SIZE - ); - if (lastNewline !== -1) { - sseBuffer = sseBuffer.slice(lastNewline + 1); - } else { - // No newline found in discard region — buffer contains an incomplete SSE line. - // Discard it entirely to avoid returning malformed data; the remainingCredits - // parser won't find valid data in a truncated line anyway. - sseBuffer = ""; - } - } - } catch { - /* decoding best-effort */ - } - }, - flush() { - // Final decode for any remaining bytes - try { - const text = decoder.decode(); // Flush pending bytes - sseBuffer += text; - } catch { - /* decoding best-effort */ - } - - // Parse the accumulated SSE data for remainingCredits - try { - const lines = sseBuffer.split("\n"); - for (const line of lines) { - const trimmed = line.trim(); - if (!trimmed.startsWith("data:")) continue; - const payload = trimmed.slice(5).trim(); - if (!payload || payload === "[DONE]") continue; - try { - const parsed = JSON.parse(payload); - if (Array.isArray(parsed?.remainingCredits)) { - const googleCredit = parsed.remainingCredits.find((c: unknown) => { - const credit = asRecord(c); - return credit?.creditType === "GOOGLE_ONE_AI"; - }) as AntigravityCreditEntry | undefined; - if (googleCredit) { - const balance = parseInt(String(googleCredit.creditAmount ?? ""), 10); - if (!isNaN(balance)) { - updateAntigravityRemainingCredits(accountId, balance); - } - } - } - } catch { - /* skip malformed lines */ - } - } - } catch { - /* credits extraction is best-effort */ - } - sseBuffer = ""; - }, - }, - { highWaterMark: 16384 }, - { highWaterMark: 16384 } + const passThrough = createCreditsExtractionTransform( + accountId, + 16 * 1024 // 16KB sliding-window cap to prevent OOM ); const tappedBody = response.body.pipeThrough(passThrough); const tappedResponse = new Response(tappedBody, { diff --git a/open-sse/handlers/chatCore/nonStreamingSse.ts b/open-sse/handlers/chatCore/nonStreamingSse.ts index 5d7e642fc62..8c2b748b978 100644 --- a/open-sse/handlers/chatCore/nonStreamingSse.ts +++ b/open-sse/handlers/chatCore/nonStreamingSse.ts @@ -3,6 +3,7 @@ import { parseSSEToResponsesOutput, parseSSEToClaudeResponse, parseSSEToOpenAIResponse, + parseSSEToGeminiResponse, } from "../sseParser.ts"; import { getHeaderValueCaseInsensitive } from "./headers.ts"; @@ -20,6 +21,7 @@ export function parseNonStreamingSSEPayload( }; queueFormat(preferredFormat); + queueFormat(FORMATS.GEMINI); queueFormat(FORMATS.OPENAI_RESPONSES); queueFormat(FORMATS.CLAUDE); queueFormat(FORMATS.OPENAI); @@ -30,7 +32,9 @@ export function parseNonStreamingSSEPayload( ? parseSSEToResponsesOutput(rawBody, fallbackModel) : format === FORMATS.CLAUDE ? parseSSEToClaudeResponse(rawBody, fallbackModel) - : parseSSEToOpenAIResponse(rawBody, fallbackModel); + : format === FORMATS.GEMINI || format === FORMATS.ANTIGRAVITY + ? parseSSEToGeminiResponse(rawBody, fallbackModel) + : parseSSEToOpenAIResponse(rawBody, fallbackModel); if (parsed && typeof parsed === "object") { return { body: parsed as Record, @@ -107,6 +111,21 @@ function hasClaudeTerminalMessageDelta(parsed: unknown, eventType: string): bool return typeof stopReason === "string" ? stopReason.length > 0 : stopReason != null; } +// Non-empty finishReason is terminal. Gemini SSE payloads from +// streamGenerateContent have candidates at the top level (no +// "response" wrapper). Any non-empty string signals stream end. +function hasGeminiTerminalFinishReason(parsed: unknown): boolean { + if (!parsed || typeof parsed !== "object") return false; + // Top-level candidates (streamGenerateContent?alt=sse) + const obj = parsed as Record; + const candidates = obj.candidates as unknown[] | undefined; + if (!Array.isArray(candidates) || candidates.length === 0) return false; + const candidate = candidates[0] as Record | undefined; + if (!candidate || typeof candidate !== "object") return false; + const finishReason = candidate.finishReason; + return typeof finishReason === "string" && finishReason.length > 0; +} + function processNonStreamingSseTerminalLine( state: NonStreamingSseTerminalState, rawLine: string @@ -130,12 +149,22 @@ function processNonStreamingSseTerminalLine( // Hot-path optimization: the terminal SSE events we look for (message_stop, // response.completed, …) all carry a top-level "type" field, OR are signalled by a - // preceding `event:` line (Claude). OpenAI chat.completion chunks carry neither and - // terminate with `[DONE]` (handled above), so parsing every one of them here is pure - // waste that compounds into the CPU-runaway on large buffered responses. Skip the - // JSON.parse unless the line could actually be a typed terminal. + // preceding `event:` line (Claude). Gemini signals completion via + // "finishReason" inside response.candidates[0]. OpenAI chat.completion chunks + // carry none of these and terminate with `[DONE]` (handled above), so parsing + // every one of them here is pure waste that compounds into the CPU-runaway on + // large buffered responses. Skip the JSON.parse unless the line could actually + // be a typed terminal. if ( !data.includes('"type"') && + // NOTE: "finishReason" is a superset match -- it triggers JSON.parse on + // every Gemini chunk that happens to contain the string (e.g. partial + // candidate payloads), not just the terminal one. This is intentional: + // the extra parses are cheap compared to the CPU-runaway we'd get from + // parsing ALL chunks unconditionally on large buffered responses, and + // the superset is safe (false positives just parse a non-terminal chunk + // and fall through to `return false`). + !data.includes('"finishReason"') && !(state.currentEvent === "message_delta" && data.includes("stop_reason")) ) { return isNonStreamingSseTerminalType(state.currentEvent); @@ -148,7 +177,9 @@ function processNonStreamingSseTerminalLine( ? parsed.type : state.currentEvent; return ( - isNonStreamingSseTerminalType(eventType) || hasClaudeTerminalMessageDelta(parsed, eventType) + isNonStreamingSseTerminalType(eventType) || + hasClaudeTerminalMessageDelta(parsed, eventType) || + hasGeminiTerminalFinishReason(parsed) ); } catch { // Keep reading malformed data so the parser can report a useful upstream error. diff --git a/open-sse/handlers/sseParser.ts b/open-sse/handlers/sseParser.ts index d2e634e12ae..b9dc848059e 100644 --- a/open-sse/handlers/sseParser.ts +++ b/open-sse/handlers/sseParser.ts @@ -1,5 +1,6 @@ import { appendToolCallArgumentDelta } from "../utils/toolCallArguments.ts"; import { sanitizeErrorMessage } from "../utils/error.ts"; +import { normalizeOpenAICompatibleFinishReasonString } from "../utils/finishReason.ts"; /** * Extract a provider error message from a buffered SSE stream that carries an @@ -823,3 +824,155 @@ export function parseSSEToResponsesOutput(rawSSE, fallbackModel) { metadata: picked.metadata || {}, }; } + +/** + * Convert Gemini/Antigravity SSE chunks into a single non-streaming OpenAI + * chat.completion JSON response. Gemini SSE carries payloads like: + * + * data: {"markdown":"...chunk..."} + * data: {"response":{"candidates":[{"content":{"parts":[{"text":"..."}]},"finishReason":"STOP"}],"usageMetadata":{...}}} + * data: {"remainingCredits":[...]} + * + * Reuses the same parsing logic as processAntigravitySSEPayload() in sseCollect.ts + * so that format conversion is functionally equivalent to the previous + * collectStreamToResponse() approach. Intentional differences: + * - remainingCredits is NOT embedded into the result (handled separately + * by the credits-extraction TransformStream in antigravity.ts). + * - The synthetic `id` uses `chatcmpl-${Date.now()}` (no UUID suffix) + * because this path runs once per response, not per chunk. + */ +export function parseSSEToGeminiResponse( + rawSSE: string, + fallbackModel: string +): Record | null { + const lines = String(rawSSE || "").split("\n"); + let textContent = ""; + let finishReason = "stop"; + let usage: Record | null = null; + let sawContent = false; + + type AccumulatedToolCall = { + id: string; + index: number; + type: "function"; + function: { name: string; arguments: string }; + }; + const toolCalls: AccumulatedToolCall[] = []; + + const stripZeroWidth = (value: unknown): unknown => { + if (typeof value === "string") return value.replace(/[\u200B-\u200D\uFEFF]/g, ""); + return value; + }; + + const tryParseTextualToolCall = (text: string): { name: string; args: unknown } | null => { + const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, ""); + const match = normalized.match( + /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/ + ); + if (!match) return null; + const name = match[1]?.trim(); + const rawArgs = match[2]?.trim(); + if (!name || !rawArgs) return null; + try { + return { name, args: stripZeroWidth(JSON.parse(rawArgs)) }; + } catch { + return null; + } + }; + + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + + try { + const parsed = JSON.parse(payload); + + // Markdown shortcut (some Gemini variants) + const markdown = + typeof parsed?.markdown === "string" + ? parsed.markdown + : typeof parsed?.response?.markdown === "string" + ? parsed.response.markdown + : null; + if (markdown) { + textContent += markdown; + sawContent = true; + } + + // Candidate content parts + const candidate = parsed?.response?.candidates?.[0]; + if (candidate?.content?.parts) { + for (const part of candidate.content.parts) { + if (typeof part.text === "string" && !part.thought && !part.thoughtSignature) { + const textualToolCall = tryParseTextualToolCall(part.text); + if (textualToolCall) { + toolCalls.push({ + id: `${textualToolCall.name}-${Date.now()}-${toolCalls.length}`, + index: toolCalls.length, + type: "function", + function: { + name: textualToolCall.name, + arguments: JSON.stringify(textualToolCall.args || {}), + }, + }); + } else { + textContent += part.text; + } + sawContent = true; + } + } + } + + if (candidate?.finishReason) { + finishReason = normalizeOpenAICompatibleFinishReasonString( + String(candidate.finishReason).toLowerCase() + ); + } + + if (parsed?.response?.usageMetadata) { + const um = parsed.response.usageMetadata; + usage = { + prompt_tokens: um.promptTokenCount || 0, + completion_tokens: um.candidatesTokenCount || 0, + total_tokens: um.totalTokenCount || 0, + }; + } + } catch { + // Ignore malformed lines + } + } + + if (!sawContent && toolCalls.length === 0) return null; + + const message: Record = { + role: "assistant", + content: textContent || null, + }; + + if (toolCalls.length > 0) { + message.tool_calls = toolCalls; + finishReason = "tool_calls"; + } + + const result: Record = { + id: `chatcmpl-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model: fallbackModel || "unknown", + choices: [ + { + index: 0, + message, + finish_reason: finishReason, + }, + ], + }; + + if (usage) { + result.usage = usage; + } + + return result; +} diff --git a/tests/unit/executor-antigravity.test.ts b/tests/unit/executor-antigravity.test.ts index 99cf20b606f..03b0063af56 100644 --- a/tests/unit/executor-antigravity.test.ts +++ b/tests/unit/executor-antigravity.test.ts @@ -1,10 +1,14 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { AntigravityExecutor } from "../../open-sse/executors/antigravity.ts"; +import { + AntigravityExecutor, + createCreditsExtractionTransform, +} from "../../open-sse/executors/antigravity.ts"; import { setCliCompatProviders } from "../../open-sse/config/cliFingerprints.ts"; import { scrubProxyAndFingerprintHeaders } from "../../open-sse/services/antigravityHeaderScrub.ts"; import { antigravityUserAgent } from "../../open-sse/services/antigravityHeaders.ts"; +import { parseSSEToGeminiResponse } from "../../open-sse/handlers/sseParser.ts"; import { clearAntigravityVersionCache, seedAntigravityVersionCache, @@ -672,7 +676,11 @@ test("AntigravityExecutor.execute auto-retries short 429 responses and collects credentials: { accessToken: "token", projectId: "project-1" }, log: { debug() {}, warn() {} }, }); - const payload = (await result.response.json()) as ChatCompletionPayload; + // Non-streaming now returns raw SSE; parse it the way chatCore would. + const rawSSE = await result.response.text(); + const parsed = parseSSEToGeminiResponse(rawSSE, "antigravity/gemini-2.5-flash"); + assert.ok(parsed, "parseSSEToGeminiResponse should parse the SSE"); + const payload = parsed as ChatCompletionPayload; assert.equal(calls.length, 2); assert.equal(result.response.status, 200); @@ -939,3 +947,111 @@ test("AntigravityExecutor.transformRequest maps Claude models through Gemini con assert.equal(result.request.temperature, undefined); assert.equal(result.request.toolConfig, undefined); }); + +// --------------------------------------------------------------------------- +// createCreditsExtractionTransform -- credits extraction with buffer cap +// --------------------------------------------------------------------------- + +test("createCreditsExtractionTransform extracts remainingCredits from SSE data", async () => { + const encoder = new TextEncoder(); + const sseData = [ + 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"hello"}]},"finishReason":"STOP"}]}}\n\n', + 'data: {"remainingCredits":[{"creditType":"GOOGLE_ONE_AI","creditAmount":"42"}]}\n\n', + ].join(""); + + const transform = createCreditsExtractionTransform("test-account"); + const readable = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(sseData)); + controller.close(); + }, + }); + + // Consume the stream through the transform + const output = readable.pipeThrough(transform); + const reader = output.getReader(); + const chunks: Uint8Array[] = []; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + chunks.push(value); + } + + // Data must pass through unmodified + const collected = new TextDecoder().decode( + new Uint8Array( + chunks.reduce((acc, c) => acc + c.length, 0) > 0 ? Buffer.concat(chunks) : new Uint8Array(0) + ) + ); + assert.ok(collected.includes("hello")); + assert.ok(collected.includes("remainingCredits")); +}); + +test("createCreditsExtractionTransform with buffer cap truncates large buffers", async () => { + const encoder = new TextEncoder(); + // Build a payload larger than 1KB + const largeText = "x".repeat(2000); + const ssePayload = JSON.stringify({ + response: { + candidates: [{ content: { parts: [{ text: largeText }] } }], + }, + }); + const sseLine = `data: ${ssePayload}\n\n`; + // Append a credits line at the end + const creditsLine = + 'data: {"remainingCredits":[{"creditType":"GOOGLE_ONE_AI","creditAmount":"99"}]}\n\n'; + const fullData = sseLine + creditsLine; + + // Use a 512-byte buffer cap -- the large text line should be discarded + const transform = createCreditsExtractionTransform("test-account", 512); + const readable = new ReadableStream({ + start(controller) { + // Send in small chunks to exercise the sliding-window logic + const encoded = encoder.encode(fullData); + const chunkSize = 256; + for (let i = 0; i < encoded.length; i += chunkSize) { + controller.enqueue(encoded.slice(i, i + chunkSize)); + } + controller.close(); + }, + }); + + const output = readable.pipeThrough(transform); + const reader = output.getReader(); + while (true) { + const { done } = await reader.read(); + if (done) break; + } + + // The transform should not throw -- buffer cap just limits what the + // flush handler can see. If the credits line was within the last 512 + // bytes it will be found; otherwise it's a graceful no-op. + // Either way, no crash or OOM. + assert.ok(true); +}); + +test("createCreditsExtractionTransform handles malformed SSE gracefully", async () => { + const encoder = new TextEncoder(); + const badData = "not valid sse\ndata: {broken json\n\ndata: [DONE]\n\n"; + + const transform = createCreditsExtractionTransform("test-account"); + const readable = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(badData)); + controller.close(); + }, + }); + + const output = readable.pipeThrough(transform); + const reader = output.getReader(); + const chunks: Uint8Array[] = []; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + chunks.push(value); + } + + // Data passes through unmodified, no crash on malformed input + const collected = new TextDecoder().decode(Buffer.concat(chunks)); + assert.ok(collected.includes("not valid sse")); +}); diff --git a/tests/unit/sse-parser.test.ts b/tests/unit/sse-parser.test.ts index 7511834df04..dc7568af5c8 100644 --- a/tests/unit/sse-parser.test.ts +++ b/tests/unit/sse-parser.test.ts @@ -1,8 +1,12 @@ import test from "node:test"; import assert from "node:assert/strict"; -const { parseSSEToOpenAIResponse, parseSSEToClaudeResponse, parseSSEToResponsesOutput } = - await import("../../open-sse/handlers/sseParser.ts"); +const { + parseSSEToOpenAIResponse, + parseSSEToClaudeResponse, + parseSSEToResponsesOutput, + parseSSEToGeminiResponse, +} = await import("../../open-sse/handlers/sseParser.ts"); test("parseSSEToOpenAIResponse parses a single SSE event with a done marker", () => { const rawSSE = [ @@ -103,12 +107,12 @@ test("parseSSEToClaudeResponse parses text, thinking, tool_use, and usage events assert.equal(parsed.id, "msg_1"); assert.equal(parsed.model, "claude-3-5-sonnet"); - assert.equal((parsed.content[0] as any).type, "thinking"); - assert.equal((parsed as any).content[0].thinking, "step 1"); - assert.equal((parsed as any).content[0].signature, "sig-1"); - (assert as any).equal((parsed.content[1] as any).text, "Hello"); - assert.equal((parsed.content[2] as any).type, "tool_use"); - (assert as any).deepEqual((parsed.content[2] as any).input, { q: "docs" }); + assert.equal((parsed.content[0] as { type: string }).type, "thinking"); + assert.equal((parsed.content[0] as { thinking: string }).thinking, "step 1"); + assert.equal((parsed.content[0] as { signature: string }).signature, "sig-1"); + assert.equal((parsed.content[1] as { text: string }).text, "Hello"); + assert.equal((parsed.content[2] as { type: string }).type, "tool_use"); + assert.deepEqual((parsed.content[2] as { input: unknown }).input, { q: "docs" }); assert.equal(parsed.stop_reason, "tool_use"); assert.equal(parsed.stop_sequence, "END"); assert.deepEqual(parsed.usage, { input_tokens: 10, output_tokens: 4 }); @@ -130,7 +134,7 @@ test("parseSSEToClaudeResponse tolerates event-only types and missing blank sepa assert.equal(parsed.id, "msg_event_fallback"); assert.equal(parsed.model, "claude-sonnet-4-6"); - assert.equal((parsed.content[0] as any).text, "event fallback ok"); + assert.equal((parsed.content[0] as { text: string }).text, "event fallback ok"); assert.deepEqual(parsed.usage, { input_tokens: 3, output_tokens: 2 }); }); @@ -152,9 +156,9 @@ test("parseSSEToClaudeResponse merges signature_delta into an existing thinking const parsed = parseSSEToClaudeResponse(rawSSE, "fallback-model"); - assert.equal((parsed.content[0] as any).type, "thinking"); - assert.equal((parsed.content[0] as any).thinking, "first second"); - assert.equal((parsed.content[0] as any).signature, "sig-1"); + assert.equal((parsed.content[0] as { type: string }).type, "thinking"); + assert.equal((parsed.content[0] as { thinking: string }).thinking, "first second"); + assert.equal((parsed.content[0] as { signature: string }).signature, "sig-1"); }); test("parseSSEToClaudeResponse preserves signature_delta when it arrives before thinking_delta", () => { @@ -172,9 +176,9 @@ test("parseSSEToClaudeResponse preserves signature_delta when it arrives before const parsed = parseSSEToClaudeResponse(rawSSE, "fallback-model"); - assert.equal((parsed.content[0] as any).type, "thinking"); - assert.equal((parsed.content[0] as any).thinking, "later thinking"); - assert.equal((parsed.content[0] as any).signature, "sig-before"); + assert.equal((parsed.content[0] as { type: string }).type, "thinking"); + assert.equal((parsed.content[0] as { thinking: string }).thinking, "later thinking"); + assert.equal((parsed.content[0] as { signature: string }).signature, "sig-before"); }); test("parseSSEToClaudeResponse ignores malformed payloads and returns null when nothing valid remains", () => { @@ -334,3 +338,98 @@ test("parseSSEToOpenAIResponse deduplicates repeated tool call snapshots", () => assert.equal(toolCall.function.arguments, args); assert.equal(JSON.parse(toolCall.function.arguments).command, "find /tmp -name test.txt"); }); + +// --------------------------------------------------------------------------- +// parseSSEToGeminiResponse +// --------------------------------------------------------------------------- + +test("parseSSEToGeminiResponse extracts text content from candidate parts", () => { + const rawSSE = [ + 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"Hello "}]}}]}}', + 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"world"}]},"finishReason":"STOP"}],"usageMetadata":{"promptTokenCount":5,"candidatesTokenCount":3,"totalTokenCount":8}}}', + ].join("\n"); + + const parsed = parseSSEToGeminiResponse(rawSSE, "gemini-2.5-flash"); + + assert.ok(parsed); + assert.equal(parsed.object, "chat.completion"); + assert.equal(parsed.choices[0].message.content, "Hello world"); + assert.equal(parsed.choices[0].finish_reason, "stop"); + assert.deepEqual(parsed.usage, { + prompt_tokens: 5, + completion_tokens: 3, + total_tokens: 8, + }); +}); + +test("parseSSEToGeminiResponse handles markdown shortcut format", () => { + const rawSSE = [ + 'data: {"markdown":"Hello "}', + 'data: {"markdown":"world"}', + 'data: {"response":{"candidates":[{"finishReason":"STOP"}]}}', + ].join("\n"); + + const parsed = parseSSEToGeminiResponse(rawSSE, "gemini-2.5-flash"); + + assert.ok(parsed); + assert.equal(parsed.choices[0].message.content, "Hello world"); + assert.equal(parsed.choices[0].finish_reason, "stop"); +}); + +test("parseSSEToGeminiResponse extracts tool calls from textual format", () => { + const rawSSE = [ + `data: ${JSON.stringify({ + response: { + candidates: [ + { + content: { + parts: [ + { + text: '[Tool call: search_files]\nArguments: {"path":"/tmp"}', + }, + ], + }, + finishReason: "STOP", + }, + ], + }, + })}`, + ].join("\n"); + + const parsed = parseSSEToGeminiResponse(rawSSE, "gemini-3.5-flash-low"); + + assert.ok(parsed); + assert.equal(parsed.choices[0].finish_reason, "tool_calls"); + const toolCalls = parsed.choices[0].message.tool_calls; + assert.equal(toolCalls.length, 1); + assert.equal(toolCalls[0].function.name, "search_files"); + assert.deepEqual(JSON.parse(toolCalls[0].function.arguments), { path: "/tmp" }); +}); + +test("parseSSEToGeminiResponse returns null for empty or non-content SSE", () => { + assert.equal(parseSSEToGeminiResponse("", "model"), null); + assert.equal(parseSSEToGeminiResponse("data: [DONE]\n", "model"), null); + assert.equal(parseSSEToGeminiResponse("event: ping\n", "model"), null); +}); + +test("parseSSEToGeminiResponse ignores thought/thoughtSignature parts", () => { + const rawSSE = [ + `data: ${JSON.stringify({ + response: { + candidates: [ + { + content: { + parts: [{ text: "internal reasoning", thought: true }, { text: "visible answer" }], + }, + finishReason: "STOP", + }, + ], + }, + })}`, + ].join("\n"); + + const parsed = parseSSEToGeminiResponse(rawSSE, "model"); + + assert.ok(parsed); + assert.equal(parsed.choices[0].message.content, "visible answer"); +}); From 846aaf9de82881d0cce4b1e1f18844c9d5f3763e Mon Sep 17 00:00:00 2001 From: HouMinXi <19586012+HouMinXi@users.noreply.github.com> Date: Sat, 18 Jul 2026 12:06:17 -0300 Subject: [PATCH 10/14] refactor(antigravity): extract streaming passthrough to module (file-size cap) #7408 added the non-streaming SSE pass-through (createCreditsExtractionTransform plus its two call sites: the credits-retry path and the main non-streaming path) inline in antigravity.ts, growing it to 1806 lines. Combined with two other authorized PRs touching the same file (#6979 +11, #7290 +30), the projected total exceeds the frozen file-size gate (1813). Extract the new streaming-passthrough logic verbatim into open-sse/executors/antigravity/streamingPassthrough.ts (createCreditsExtractionTransform + a new buildSsePassthroughResult that deduplicates the two near-identical call sites), following the existing sseCollect.ts submodule pattern -- pure, no host state, no fetch/auth. antigravity.ts keeps a thin wrapper for createCreditsExtractionTransform (same public signature the existing unit tests import) that injects updateAntigravityRemainingCredits so the two modules don't import each other. No behavior change: same abort handling, same 499-on-early-disconnect, same 16KB credits sliding-window cap. antigravity.ts: 1806 -> 1693 lines (under the 1755 pre-PR baseline, with margin). New module: 176 lines (cap 800). Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> --- open-sse/executors/antigravity.ts | 185 ++++-------------- .../antigravity/streamingPassthrough.ts | 176 +++++++++++++++++ 2 files changed, 212 insertions(+), 149 deletions(-) create mode 100644 open-sse/executors/antigravity/streamingPassthrough.ts diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index dc82426a33c..bd87c2f7dd0 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -54,6 +54,10 @@ import { } from "./antigravity/sseCollect.ts"; // processAntigravitySSEPayload re-exported for external importers (tests). export { processAntigravitySSEPayload } from "./antigravity/sseCollect.ts"; +import { + createCreditsExtractionTransform as createCreditsExtractionTransformImpl, + buildSsePassthroughResult, +} from "./antigravity/streamingPassthrough.ts"; import { applyAntigravityClientProfileHeaders, removeHeaderCaseInsensitive, @@ -120,11 +124,6 @@ type AntigravityChunkContent = Record & { >; }; -type AntigravityCreditEntry = { - creditType?: string; - creditAmount?: string; -}; - function getChunkedOrFixedBody(bodyStr: string, stream: boolean): BodyInit { if (stream) { return new ReadableStream( @@ -273,87 +272,22 @@ export function updateAntigravityRemainingCredits(accountId: string, balance: nu } /** - * Create a pass-through TransformStream that extracts `remainingCredits` - * from SSE data without consuming the stream. The downstream client - * receives the unmodified bytes. - * - * @param accountId Provider account ID for credit-balance persistence. - * @param bufferSize Optional sliding-window buffer cap in bytes. - * Pass 0 or omit for unlimited (non-streaming callers - * where the full body is already buffered upstream). - * The streaming path uses 16384 (16 KB) to prevent OOM - * on long-lived SSE connections. Credit-balance data - * appears near the end of the SSE stream (after - * content), so the sliding window captures it even at - * 16 KB -- only truly massive responses (>16 KB of - * consecutive non-newline content) would lose credits. + * Pass-through TransformStream that extracts `remainingCredits` from SSE + * data without consuming the stream (the downstream client receives the + * unmodified bytes). Thin wrapper around the pure implementation in + * streamingPassthrough.ts, injecting this executor's credit-balance cache + * writer so the two modules don't import each other. See that module's + * doc comment for the full parameter behavior. * @internal Exported for unit testing only. */ export function createCreditsExtractionTransform( accountId: string, bufferSize = 0 ): TransformStream { - let buffer = ""; - const decoder = new TextDecoder(); - - return new TransformStream( - { - transform(chunk, controller) { - controller.enqueue(chunk); - try { - buffer += decoder.decode(chunk, { stream: true }); - // Sliding-window cap: truncate after the last complete newline - // in the discard region so SSE lines are never split mid-payload. - if (bufferSize > 0 && buffer.length > bufferSize) { - const lastNewline = buffer.lastIndexOf("\n", buffer.length - bufferSize); - if (lastNewline !== -1) { - buffer = buffer.slice(lastNewline + 1); - } else { - // No newline in the discard region -- incomplete line, discard entirely. - buffer = ""; - } - } - } catch { - /* decoding best-effort */ - } - }, - flush() { - try { - buffer += decoder.decode(); - } catch { - /* decoding best-effort */ - } - try { - const lines = buffer.split("\n"); - for (const line of lines) { - const trimmed = line.trim(); - if (!trimmed.startsWith("data:")) continue; - const payload = trimmed.slice(5).trim(); - if (!payload || payload === "[DONE]") continue; - try { - const parsed = JSON.parse(payload); - if (Array.isArray(parsed?.remainingCredits)) { - const googleCredit = parsed.remainingCredits.find((c: unknown) => { - const credit = asRecord(c); - return credit?.creditType === "GOOGLE_ONE_AI"; - }) as AntigravityCreditEntry | undefined; - if (googleCredit) { - const balance = parseInt(String(googleCredit.creditAmount ?? ""), 10); - if (!isNaN(balance)) updateAntigravityRemainingCredits(accountId, balance); - } - } - } catch { - /* skip malformed lines */ - } - } - } catch { - /* credits extraction is best-effort */ - } - buffer = ""; - }, - }, - { highWaterMark: 16384 }, - { highWaterMark: 16384 } + return createCreditsExtractionTransformImpl( + accountId, + updateAntigravityRemainingCredits, + bufferSize ); } @@ -1465,41 +1399,19 @@ export class AntigravityExecutor extends BaseExecutor { if (creditsResp.ok || creditsResp.status !== HTTP_STATUS.RATE_LIMITED) { log?.info?.("AG_CREDITS", `Credits retry succeeded: ${creditsResp.status}`); if (!stream && creditsResp.body) { - // Client already disconnected — skip pipe - if (signal?.aborted) { - creditsResp.body.cancel().catch(() => {}); - return { - response: new Response(null, { status: 499 }), - url, - headers: finalCreditsHeaders, - transformedBody: null, - }; - } - // Return raw SSE with credits extraction - const crTransform = createCreditsExtractionTransform( + // Raw SSE pass-through + credits extraction (see + // streamingPassthrough.ts); 499s early if the client + // already disconnected instead of piping a cancelled body. + return buildSsePassthroughResult( + creditsResp.body, + creditsResp, accountId, - 16 * 1024 // 16KB cap - ); - const crTappedBody = creditsResp.body.pipeThrough(crTransform); - if (signal) { - signal.addEventListener( - "abort", - () => { - creditsResp.body.cancel().catch(() => {}); - }, - { once: true } - ); - } - return { - response: new Response(crTappedBody, { - status: creditsResp.status, - statusText: creditsResp.statusText, - headers: creditsResp.headers, - }), + updateAntigravityRemainingCredits, url, - headers: finalCreditsHeaders, - transformedBody: attachToolNameMap(creditsBody, requestToolNameMap), - }; + finalCreditsHeaders, + attachToolNameMap(creditsBody, requestToolNameMap), + signal + ); } return { response: creditsResp, @@ -1665,44 +1577,19 @@ export class AntigravityExecutor extends BaseExecutor { } if (response.body) { - // Client already disconnected — skip pipe - if (signal?.aborted) { - response.body.cancel().catch(() => {}); - return { - response: new Response(null, { status: 499 }), - url, - headers: finalHeaders, - transformedBody: null, - }; - } - // Cancel upstream body on client disconnect - if (signal) { - signal.addEventListener( - "abort", - () => { - response.body?.cancel().catch(() => {}); - }, - { once: true } - ); - } - - // Tap the stream to extract remainingCredits while passing - // data through unmodified. chatCore drains the full body. - const nsPassThrough = createCreditsExtractionTransform( + // Raw SSE pass-through + credits extraction (see + // streamingPassthrough.ts); 499s early if the client already + // disconnected instead of piping a cancelled body. + return buildSsePassthroughResult( + response.body, + response, accountId, - 16 * 1024 // 16KB sliding-window cap to prevent OOM - ); - const tappedBody = response.body.pipeThrough(nsPassThrough); - return { - response: new Response(tappedBody, { - status: response.status, - statusText: response.statusText, - headers: response.headers, - }), + updateAntigravityRemainingCredits, url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; + finalHeaders, + attachToolNameMap(transformedBody, requestToolNameMap), + signal + ); } // No body -- return as-is diff --git a/open-sse/executors/antigravity/streamingPassthrough.ts b/open-sse/executors/antigravity/streamingPassthrough.ts new file mode 100644 index 00000000000..69282e030c3 --- /dev/null +++ b/open-sse/executors/antigravity/streamingPassthrough.ts @@ -0,0 +1,176 @@ +// Pure streaming pass-through helpers for the Antigravity executor (#7408): +// tap an upstream Gemini SSE Response through a credits-extraction +// TransformStream instead of buffering the whole body in the executor, so +// long-thinking models aren't killed by an artificial collection timeout. +// Extracted from antigravity.ts (no host state, no fetch/auth) -- the +// credit-balance cache itself stays in antigravity.ts; callers inject the +// update function below so the two modules don't import each other. + +/** Shape of one entry in a Gemini `remainingCredits` SSE payload array. */ +export type AntigravityCreditEntry = { + creditType?: string; + creditAmount?: string; +}; + +function asCreditRecord(value: unknown): Record | null { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : null; +} + +/** + * Create a pass-through TransformStream that extracts `remainingCredits` + * from SSE data without consuming the stream. The downstream client + * receives the unmodified bytes. + * + * @param accountId Provider account ID for credit-balance persistence. + * @param onCreditsUpdate Invoked with the parsed GOOGLE_ONE_AI balance. + * Injected by the caller (antigravity.ts's + * updateAntigravityRemainingCredits) to avoid this module + * importing back the executor's credit-balance cache. + * @param bufferSize Optional sliding-window buffer cap in bytes. + * Pass 0 or omit for unlimited (non-streaming callers + * where the full body is already buffered upstream). + * The streaming path uses 16384 (16 KB) to prevent OOM + * on long-lived SSE connections. Credit-balance data + * appears near the end of the SSE stream (after + * content), so the sliding window captures it even at + * 16 KB -- only truly massive responses (>16 KB of + * consecutive non-newline content) would lose credits. + */ +export function createCreditsExtractionTransform( + accountId: string, + onCreditsUpdate: (accountId: string, balance: number) => void, + bufferSize = 0 +): TransformStream { + let buffer = ""; + const decoder = new TextDecoder(); + + return new TransformStream( + { + transform(chunk, controller) { + controller.enqueue(chunk); + try { + buffer += decoder.decode(chunk, { stream: true }); + // Sliding-window cap: truncate after the last complete newline + // in the discard region so SSE lines are never split mid-payload. + if (bufferSize > 0 && buffer.length > bufferSize) { + const lastNewline = buffer.lastIndexOf("\n", buffer.length - bufferSize); + if (lastNewline !== -1) { + buffer = buffer.slice(lastNewline + 1); + } else { + // No newline in the discard region -- incomplete line, discard entirely. + buffer = ""; + } + } + } catch { + /* decoding best-effort */ + } + }, + flush() { + try { + buffer += decoder.decode(); + } catch { + /* decoding best-effort */ + } + try { + const lines = buffer.split("\n"); + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + try { + const parsed = JSON.parse(payload); + if (Array.isArray(parsed?.remainingCredits)) { + const googleCredit = parsed.remainingCredits.find((c: unknown) => { + const credit = asCreditRecord(c); + return credit?.creditType === "GOOGLE_ONE_AI"; + }) as AntigravityCreditEntry | undefined; + if (googleCredit) { + const balance = parseInt(String(googleCredit.creditAmount ?? ""), 10); + if (!isNaN(balance)) onCreditsUpdate(accountId, balance); + } + } + } catch { + /* skip malformed lines */ + } + } + } catch { + /* credits extraction is best-effort */ + } + buffer = ""; + }, + }, + { highWaterMark: 16384 }, + { highWaterMark: 16384 } + ); +} + +/** Result shape returned to callers of AntigravityExecutor.execute(). */ +export type SsePassthroughResult = { + response: Response; + url: string; + headers: Record; + transformedBody: unknown; +}; + +/** Cancel `body` when `signal` aborts, releasing the upstream connection. */ +function cancelBodyOnAbort(body: ReadableStream, signal: AbortSignal): void { + signal.addEventListener( + "abort", + () => { + body.cancel().catch(() => {}); + }, + { once: true } + ); +} + +/** + * Build the non-streaming pass-through result: tap `body` through + * createCreditsExtractionTransform and wrap it in a same-status Response so + * chatCore's non-streaming path (readNonStreamingResponseBody + + * parseNonStreamingSSEPayload) can drain and parse the Gemini SSE without + * this executor buffering the whole stream itself. + * + * If the client already disconnected (`signal.aborted`), cancels the + * upstream body immediately and returns a bare 499 instead of piping a + * cancelled body through. + */ +export function buildSsePassthroughResult( + body: ReadableStream, + upstream: { status: number; statusText: string; headers: Headers }, + accountId: string, + onCreditsUpdate: (accountId: string, balance: number) => void, + url: string, + outHeaders: Record, + transformedBody: unknown, + signal: AbortSignal | null | undefined +): SsePassthroughResult { + // Client already disconnected — skip pipe + if (signal?.aborted) { + body.cancel().catch(() => {}); + return { + response: new Response(null, { status: 499 }), + url, + headers: outHeaders, + transformedBody: null, + }; + } + // Cancel upstream body on client disconnect + if (signal) cancelBodyOnAbort(body, signal); + + const tapped = body.pipeThrough( + createCreditsExtractionTransform(accountId, onCreditsUpdate, 16 * 1024) + ); + return { + response: new Response(tapped, { + status: upstream.status, + statusText: upstream.statusText, + headers: upstream.headers, + }), + url, + headers: outHeaders, + transformedBody, + }; +} From 3180d89b460df148d0176f0b24b821bb9ad8e575 Mon Sep 17 00:00:00 2001 From: HouMinXi <19586012+HouMinXi@users.noreply.github.com> Date: Sat, 18 Jul 2026 12:19:18 -0300 Subject: [PATCH 11/14] refactor(antigravity): split incremental parser + move new tests to own file (file-size caps) Two remaining frozen file-size violations from #7408, resolved by extraction/move with zero behavior or assert changes: - open-sse/handlers/sseParser.ts (979 > frozen 830): the PR appended parseSSEToGeminiResponse (+153, the Gemini buffered-SSE -> chat.completion parser). Moved verbatim to open-sse/handlers/sseParser/geminiResponse.ts, following the handlers submodule pattern (chatCore/, responseSanitizer/). sseParser.ts is now byte-identical to its pre-PR content (825 lines; PR delta 0). Importers (chatCore/nonStreamingSse.ts, tests) point at the new module. - tests/unit/executor-antigravity.test.ts (1058 > testFrozen 942): the PR's new streaming-passthrough tests moved verbatim (same tests, same asserts) to tests/unit/antigravity-streaming-passthrough.test.ts: the 3 createCreditsExtractionTransform tests plus the non-streaming passthrough drain test ("auto-retries short 429 ... collects SSE for non-stream clients"), which the PR rewired onto the new raw-SSE path. The frozen file drops to 888 lines (below its pre-PR 941). New files: geminiResponse.ts 156 lines, passthrough test 202 lines (caps 800). Also fixes the stale sseParser.ts path in collectStreamToResponse's deprecation note. Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> --- open-sse/executors/antigravity.ts | 4 +- open-sse/handlers/chatCore/nonStreamingSse.ts | 2 +- open-sse/handlers/sseParser.ts | 153 ------------- open-sse/handlers/sseParser/geminiResponse.ts | 156 ++++++++++++++ .../antigravity-streaming-passthrough.test.ts | 202 ++++++++++++++++++ tests/unit/executor-antigravity.test.ts | 177 +-------------- tests/unit/sse-parser.test.ts | 10 +- 7 files changed, 369 insertions(+), 335 deletions(-) create mode 100644 open-sse/handlers/sseParser/geminiResponse.ts create mode 100644 tests/unit/antigravity-streaming-passthrough.test.ts diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index bd87c2f7dd0..283cd6689c0 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -952,8 +952,8 @@ export class AntigravityExecutor extends BaseExecutor { * OpenAI-format chat.completion payload. * * @deprecated Use the non-streaming SSE path in chatCore instead, which calls - * parseSSEToGeminiResponse() from sseParser.ts. This method is retained only - * for backward compatibility and may be removed in a future release. + * parseSSEToGeminiResponse() from sseParser/geminiResponse.ts. This method is + * retained only for backward compatibility and may be removed in a future release. */ collectStreamToResponse( response: Response, diff --git a/open-sse/handlers/chatCore/nonStreamingSse.ts b/open-sse/handlers/chatCore/nonStreamingSse.ts index 8c2b748b978..743ba81a918 100644 --- a/open-sse/handlers/chatCore/nonStreamingSse.ts +++ b/open-sse/handlers/chatCore/nonStreamingSse.ts @@ -3,8 +3,8 @@ import { parseSSEToResponsesOutput, parseSSEToClaudeResponse, parseSSEToOpenAIResponse, - parseSSEToGeminiResponse, } from "../sseParser.ts"; +import { parseSSEToGeminiResponse } from "../sseParser/geminiResponse.ts"; import { getHeaderValueCaseInsensitive } from "./headers.ts"; export function parseNonStreamingSSEPayload( diff --git a/open-sse/handlers/sseParser.ts b/open-sse/handlers/sseParser.ts index b9dc848059e..d2e634e12ae 100644 --- a/open-sse/handlers/sseParser.ts +++ b/open-sse/handlers/sseParser.ts @@ -1,6 +1,5 @@ import { appendToolCallArgumentDelta } from "../utils/toolCallArguments.ts"; import { sanitizeErrorMessage } from "../utils/error.ts"; -import { normalizeOpenAICompatibleFinishReasonString } from "../utils/finishReason.ts"; /** * Extract a provider error message from a buffered SSE stream that carries an @@ -824,155 +823,3 @@ export function parseSSEToResponsesOutput(rawSSE, fallbackModel) { metadata: picked.metadata || {}, }; } - -/** - * Convert Gemini/Antigravity SSE chunks into a single non-streaming OpenAI - * chat.completion JSON response. Gemini SSE carries payloads like: - * - * data: {"markdown":"...chunk..."} - * data: {"response":{"candidates":[{"content":{"parts":[{"text":"..."}]},"finishReason":"STOP"}],"usageMetadata":{...}}} - * data: {"remainingCredits":[...]} - * - * Reuses the same parsing logic as processAntigravitySSEPayload() in sseCollect.ts - * so that format conversion is functionally equivalent to the previous - * collectStreamToResponse() approach. Intentional differences: - * - remainingCredits is NOT embedded into the result (handled separately - * by the credits-extraction TransformStream in antigravity.ts). - * - The synthetic `id` uses `chatcmpl-${Date.now()}` (no UUID suffix) - * because this path runs once per response, not per chunk. - */ -export function parseSSEToGeminiResponse( - rawSSE: string, - fallbackModel: string -): Record | null { - const lines = String(rawSSE || "").split("\n"); - let textContent = ""; - let finishReason = "stop"; - let usage: Record | null = null; - let sawContent = false; - - type AccumulatedToolCall = { - id: string; - index: number; - type: "function"; - function: { name: string; arguments: string }; - }; - const toolCalls: AccumulatedToolCall[] = []; - - const stripZeroWidth = (value: unknown): unknown => { - if (typeof value === "string") return value.replace(/[\u200B-\u200D\uFEFF]/g, ""); - return value; - }; - - const tryParseTextualToolCall = (text: string): { name: string; args: unknown } | null => { - const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, ""); - const match = normalized.match( - /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/ - ); - if (!match) return null; - const name = match[1]?.trim(); - const rawArgs = match[2]?.trim(); - if (!name || !rawArgs) return null; - try { - return { name, args: stripZeroWidth(JSON.parse(rawArgs)) }; - } catch { - return null; - } - }; - - for (const line of lines) { - const trimmed = line.trim(); - if (!trimmed.startsWith("data:")) continue; - const payload = trimmed.slice(5).trim(); - if (!payload || payload === "[DONE]") continue; - - try { - const parsed = JSON.parse(payload); - - // Markdown shortcut (some Gemini variants) - const markdown = - typeof parsed?.markdown === "string" - ? parsed.markdown - : typeof parsed?.response?.markdown === "string" - ? parsed.response.markdown - : null; - if (markdown) { - textContent += markdown; - sawContent = true; - } - - // Candidate content parts - const candidate = parsed?.response?.candidates?.[0]; - if (candidate?.content?.parts) { - for (const part of candidate.content.parts) { - if (typeof part.text === "string" && !part.thought && !part.thoughtSignature) { - const textualToolCall = tryParseTextualToolCall(part.text); - if (textualToolCall) { - toolCalls.push({ - id: `${textualToolCall.name}-${Date.now()}-${toolCalls.length}`, - index: toolCalls.length, - type: "function", - function: { - name: textualToolCall.name, - arguments: JSON.stringify(textualToolCall.args || {}), - }, - }); - } else { - textContent += part.text; - } - sawContent = true; - } - } - } - - if (candidate?.finishReason) { - finishReason = normalizeOpenAICompatibleFinishReasonString( - String(candidate.finishReason).toLowerCase() - ); - } - - if (parsed?.response?.usageMetadata) { - const um = parsed.response.usageMetadata; - usage = { - prompt_tokens: um.promptTokenCount || 0, - completion_tokens: um.candidatesTokenCount || 0, - total_tokens: um.totalTokenCount || 0, - }; - } - } catch { - // Ignore malformed lines - } - } - - if (!sawContent && toolCalls.length === 0) return null; - - const message: Record = { - role: "assistant", - content: textContent || null, - }; - - if (toolCalls.length > 0) { - message.tool_calls = toolCalls; - finishReason = "tool_calls"; - } - - const result: Record = { - id: `chatcmpl-${Date.now()}`, - object: "chat.completion", - created: Math.floor(Date.now() / 1000), - model: fallbackModel || "unknown", - choices: [ - { - index: 0, - message, - finish_reason: finishReason, - }, - ], - }; - - if (usage) { - result.usage = usage; - } - - return result; -} diff --git a/open-sse/handlers/sseParser/geminiResponse.ts b/open-sse/handlers/sseParser/geminiResponse.ts new file mode 100644 index 00000000000..d0801019bd2 --- /dev/null +++ b/open-sse/handlers/sseParser/geminiResponse.ts @@ -0,0 +1,156 @@ +// Gemini/Antigravity buffered-SSE -> chat.completion conversion (#7408). +// Extracted verbatim from sseParser.ts (file-size cap): pure parsing, no host +// state, following the handlers submodule pattern (chatCore/, responseSanitizer/). +import { normalizeOpenAICompatibleFinishReasonString } from "../../utils/finishReason.ts"; + +/** + * Convert Gemini/Antigravity SSE chunks into a single non-streaming OpenAI + * chat.completion JSON response. Gemini SSE carries payloads like: + * + * data: {"markdown":"...chunk..."} + * data: {"response":{"candidates":[{"content":{"parts":[{"text":"..."}]},"finishReason":"STOP"}],"usageMetadata":{...}}} + * data: {"remainingCredits":[...]} + * + * Reuses the same parsing logic as processAntigravitySSEPayload() in sseCollect.ts + * so that format conversion is functionally equivalent to the previous + * collectStreamToResponse() approach. Intentional differences: + * - remainingCredits is NOT embedded into the result (handled separately + * by the credits-extraction TransformStream in antigravity.ts). + * - The synthetic `id` uses `chatcmpl-${Date.now()}` (no UUID suffix) + * because this path runs once per response, not per chunk. + */ +export function parseSSEToGeminiResponse( + rawSSE: string, + fallbackModel: string +): Record | null { + const lines = String(rawSSE || "").split("\n"); + let textContent = ""; + let finishReason = "stop"; + let usage: Record | null = null; + let sawContent = false; + + type AccumulatedToolCall = { + id: string; + index: number; + type: "function"; + function: { name: string; arguments: string }; + }; + const toolCalls: AccumulatedToolCall[] = []; + + const stripZeroWidth = (value: unknown): unknown => { + if (typeof value === "string") return value.replace(/[\u200B-\u200D\uFEFF]/g, ""); + return value; + }; + + const tryParseTextualToolCall = (text: string): { name: string; args: unknown } | null => { + const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, ""); + const match = normalized.match( + /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/ + ); + if (!match) return null; + const name = match[1]?.trim(); + const rawArgs = match[2]?.trim(); + if (!name || !rawArgs) return null; + try { + return { name, args: stripZeroWidth(JSON.parse(rawArgs)) }; + } catch { + return null; + } + }; + + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + + try { + const parsed = JSON.parse(payload); + + // Markdown shortcut (some Gemini variants) + const markdown = + typeof parsed?.markdown === "string" + ? parsed.markdown + : typeof parsed?.response?.markdown === "string" + ? parsed.response.markdown + : null; + if (markdown) { + textContent += markdown; + sawContent = true; + } + + // Candidate content parts + const candidate = parsed?.response?.candidates?.[0]; + if (candidate?.content?.parts) { + for (const part of candidate.content.parts) { + if (typeof part.text === "string" && !part.thought && !part.thoughtSignature) { + const textualToolCall = tryParseTextualToolCall(part.text); + if (textualToolCall) { + toolCalls.push({ + id: `${textualToolCall.name}-${Date.now()}-${toolCalls.length}`, + index: toolCalls.length, + type: "function", + function: { + name: textualToolCall.name, + arguments: JSON.stringify(textualToolCall.args || {}), + }, + }); + } else { + textContent += part.text; + } + sawContent = true; + } + } + } + + if (candidate?.finishReason) { + finishReason = normalizeOpenAICompatibleFinishReasonString( + String(candidate.finishReason).toLowerCase() + ); + } + + if (parsed?.response?.usageMetadata) { + const um = parsed.response.usageMetadata; + usage = { + prompt_tokens: um.promptTokenCount || 0, + completion_tokens: um.candidatesTokenCount || 0, + total_tokens: um.totalTokenCount || 0, + }; + } + } catch { + // Ignore malformed lines + } + } + + if (!sawContent && toolCalls.length === 0) return null; + + const message: Record = { + role: "assistant", + content: textContent || null, + }; + + if (toolCalls.length > 0) { + message.tool_calls = toolCalls; + finishReason = "tool_calls"; + } + + const result: Record = { + id: `chatcmpl-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model: fallbackModel || "unknown", + choices: [ + { + index: 0, + message, + finish_reason: finishReason, + }, + ], + }; + + if (usage) { + result.usage = usage; + } + + return result; +} diff --git a/tests/unit/antigravity-streaming-passthrough.test.ts b/tests/unit/antigravity-streaming-passthrough.test.ts new file mode 100644 index 00000000000..23715cbae30 --- /dev/null +++ b/tests/unit/antigravity-streaming-passthrough.test.ts @@ -0,0 +1,202 @@ +// Antigravity streaming-passthrough behavior (#7408): the non-streaming drain +// path (raw SSE + chatCore-side parse) and the credits-extraction pass-through +// TransformStream. Moved verbatim from tests/unit/executor-antigravity.test.ts +// (frozen file-size cap) — same tests, same asserts. +import { test } from "node:test"; +import assert from "node:assert/strict"; + +import { + AntigravityExecutor, + createCreditsExtractionTransform, +} from "../../open-sse/executors/antigravity.ts"; +import { parseSSEToGeminiResponse } from "../../open-sse/handlers/sseParser/geminiResponse.ts"; +import { + clearAntigravityVersionCache, + seedAntigravityVersionCache, +} from "../../open-sse/services/antigravityVersion.ts"; + +type ChatCompletionPayload = { + object?: string; + choices: Array<{ + message: { content: string }; + finish_reason: string; + }>; + usage?: { + prompt_tokens: number; + completion_tokens: number; + total_tokens: number; + }; +}; + +test.afterEach(() => { + clearAntigravityVersionCache(); +}); + +test("AntigravityExecutor.execute auto-retries short 429 responses and collects SSE for non-stream clients", async () => { + const executor = new AntigravityExecutor(); + const originalFetch = globalThis.fetch; + const originalSetTimeout = globalThis.setTimeout; + const calls = []; + seedAntigravityVersionCache("2026.04.17-test"); + + globalThis.fetch = async (url) => { + calls.push(String(url)); + + if (calls.length === 1) { + return new Response(JSON.stringify({ error: { message: "rate limited" } }), { + status: 429, + headers: { "Content-Type": "application/json" }, + }); + } + + return new Response( + [ + 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"Hello "}]},"finishReason":"STOP"}]}}\n\n', + 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"again"}]},"finishReason":"STOP"}],"usageMetadata":{"promptTokenCount":2,"candidatesTokenCount":3,"totalTokenCount":5}}}\n\n', + ].join(""), + { + status: 200, + headers: { "Content-Type": "text/event-stream" }, + } + ); + }; + globalThis.setTimeout = ((callback) => { + (callback as () => void)(); + return 0; + }) as typeof setTimeout; + + try { + const result = await executor.execute({ + model: "antigravity/gemini-2.5-flash", + body: { request: { contents: [] } }, + stream: false, + credentials: { accessToken: "token", projectId: "project-1" }, + log: { debug() {}, warn() {} }, + }); + // Non-streaming now returns raw SSE; parse it the way chatCore would. + const rawSSE = await result.response.text(); + const parsed = parseSSEToGeminiResponse(rawSSE, "antigravity/gemini-2.5-flash"); + assert.ok(parsed, "parseSSEToGeminiResponse should parse the SSE"); + const payload = parsed as ChatCompletionPayload; + + assert.equal(calls.length, 2); + assert.equal(result.response.status, 200); + assert.equal(payload.choices[0].message.content, "Hello again"); + assert.deepEqual(payload.usage, { + prompt_tokens: 2, + completion_tokens: 3, + total_tokens: 5, + }); + } finally { + globalThis.fetch = originalFetch; + globalThis.setTimeout = originalSetTimeout; + } +}); + +// --------------------------------------------------------------------------- +// createCreditsExtractionTransform -- credits extraction with buffer cap +// --------------------------------------------------------------------------- + +test("createCreditsExtractionTransform extracts remainingCredits from SSE data", async () => { + const encoder = new TextEncoder(); + const sseData = [ + 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"hello"}]},"finishReason":"STOP"}]}}\n\n', + 'data: {"remainingCredits":[{"creditType":"GOOGLE_ONE_AI","creditAmount":"42"}]}\n\n', + ].join(""); + + const transform = createCreditsExtractionTransform("test-account"); + const readable = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(sseData)); + controller.close(); + }, + }); + + // Consume the stream through the transform + const output = readable.pipeThrough(transform); + const reader = output.getReader(); + const chunks: Uint8Array[] = []; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + chunks.push(value); + } + + // Data must pass through unmodified + const collected = new TextDecoder().decode( + new Uint8Array( + chunks.reduce((acc, c) => acc + c.length, 0) > 0 ? Buffer.concat(chunks) : new Uint8Array(0) + ) + ); + assert.ok(collected.includes("hello")); + assert.ok(collected.includes("remainingCredits")); +}); + +test("createCreditsExtractionTransform with buffer cap truncates large buffers", async () => { + const encoder = new TextEncoder(); + // Build a payload larger than 1KB + const largeText = "x".repeat(2000); + const ssePayload = JSON.stringify({ + response: { + candidates: [{ content: { parts: [{ text: largeText }] } }], + }, + }); + const sseLine = `data: ${ssePayload}\n\n`; + // Append a credits line at the end + const creditsLine = + 'data: {"remainingCredits":[{"creditType":"GOOGLE_ONE_AI","creditAmount":"99"}]}\n\n'; + const fullData = sseLine + creditsLine; + + // Use a 512-byte buffer cap -- the large text line should be discarded + const transform = createCreditsExtractionTransform("test-account", 512); + const readable = new ReadableStream({ + start(controller) { + // Send in small chunks to exercise the sliding-window logic + const encoded = encoder.encode(fullData); + const chunkSize = 256; + for (let i = 0; i < encoded.length; i += chunkSize) { + controller.enqueue(encoded.slice(i, i + chunkSize)); + } + controller.close(); + }, + }); + + const output = readable.pipeThrough(transform); + const reader = output.getReader(); + while (true) { + const { done } = await reader.read(); + if (done) break; + } + + // The transform should not throw -- buffer cap just limits what the + // flush handler can see. If the credits line was within the last 512 + // bytes it will be found; otherwise it's a graceful no-op. + // Either way, no crash or OOM. + assert.ok(true); +}); + +test("createCreditsExtractionTransform handles malformed SSE gracefully", async () => { + const encoder = new TextEncoder(); + const badData = "not valid sse\ndata: {broken json\n\ndata: [DONE]\n\n"; + + const transform = createCreditsExtractionTransform("test-account"); + const readable = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(badData)); + controller.close(); + }, + }); + + const output = readable.pipeThrough(transform); + const reader = output.getReader(); + const chunks: Uint8Array[] = []; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + chunks.push(value); + } + + // Data passes through unmodified, no crash on malformed input + const collected = new TextDecoder().decode(Buffer.concat(chunks)); + assert.ok(collected.includes("not valid sse")); +}); diff --git a/tests/unit/executor-antigravity.test.ts b/tests/unit/executor-antigravity.test.ts index 03b0063af56..d92db9fb8a0 100644 --- a/tests/unit/executor-antigravity.test.ts +++ b/tests/unit/executor-antigravity.test.ts @@ -1,14 +1,10 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { - AntigravityExecutor, - createCreditsExtractionTransform, -} from "../../open-sse/executors/antigravity.ts"; +import { AntigravityExecutor } from "../../open-sse/executors/antigravity.ts"; import { setCliCompatProviders } from "../../open-sse/config/cliFingerprints.ts"; import { scrubProxyAndFingerprintHeaders } from "../../open-sse/services/antigravityHeaderScrub.ts"; import { antigravityUserAgent } from "../../open-sse/services/antigravityHeaders.ts"; -import { parseSSEToGeminiResponse } from "../../open-sse/handlers/sseParser.ts"; import { clearAntigravityVersionCache, seedAntigravityVersionCache, @@ -635,66 +631,9 @@ test("AntigravityExecutor.refreshCredentials refreshes Google OAuth tokens", asy } }); -test("AntigravityExecutor.execute auto-retries short 429 responses and collects SSE for non-stream clients", async () => { - const executor = new AntigravityExecutor(); - const originalFetch = globalThis.fetch; - const originalSetTimeout = globalThis.setTimeout; - const calls = []; - seedAntigravityVersionCache("2026.04.17-test"); - - globalThis.fetch = async (url) => { - calls.push(String(url)); - - if (calls.length === 1) { - return new Response(JSON.stringify({ error: { message: "rate limited" } }), { - status: 429, - headers: { "Content-Type": "application/json" }, - }); - } - - return new Response( - [ - 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"Hello "}]},"finishReason":"STOP"}]}}\n\n', - 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"again"}]},"finishReason":"STOP"}],"usageMetadata":{"promptTokenCount":2,"candidatesTokenCount":3,"totalTokenCount":5}}}\n\n', - ].join(""), - { - status: 200, - headers: { "Content-Type": "text/event-stream" }, - } - ); - }; - globalThis.setTimeout = ((callback) => { - (callback as () => void)(); - return 0; - }) as typeof setTimeout; - - try { - const result = await executor.execute({ - model: "antigravity/gemini-2.5-flash", - body: { request: { contents: [] } }, - stream: false, - credentials: { accessToken: "token", projectId: "project-1" }, - log: { debug() {}, warn() {} }, - }); - // Non-streaming now returns raw SSE; parse it the way chatCore would. - const rawSSE = await result.response.text(); - const parsed = parseSSEToGeminiResponse(rawSSE, "antigravity/gemini-2.5-flash"); - assert.ok(parsed, "parseSSEToGeminiResponse should parse the SSE"); - const payload = parsed as ChatCompletionPayload; - - assert.equal(calls.length, 2); - assert.equal(result.response.status, 200); - assert.equal(payload.choices[0].message.content, "Hello again"); - assert.deepEqual(payload.usage, { - prompt_tokens: 2, - completion_tokens: 3, - total_tokens: 5, - }); - } finally { - globalThis.fetch = originalFetch; - globalThis.setTimeout = originalSetTimeout; - } -}); +// The non-streaming passthrough drain test ("auto-retries short 429 responses and +// collects SSE for non-stream clients") lives in +// tests/unit/antigravity-streaming-passthrough.test.ts with the other passthrough tests. test("AntigravityExecutor.execute embeds retryAfterMs when the upstream asks for a long wait", async () => { const executor = new AntigravityExecutor(); @@ -947,111 +886,3 @@ test("AntigravityExecutor.transformRequest maps Claude models through Gemini con assert.equal(result.request.temperature, undefined); assert.equal(result.request.toolConfig, undefined); }); - -// --------------------------------------------------------------------------- -// createCreditsExtractionTransform -- credits extraction with buffer cap -// --------------------------------------------------------------------------- - -test("createCreditsExtractionTransform extracts remainingCredits from SSE data", async () => { - const encoder = new TextEncoder(); - const sseData = [ - 'data: {"response":{"candidates":[{"content":{"parts":[{"text":"hello"}]},"finishReason":"STOP"}]}}\n\n', - 'data: {"remainingCredits":[{"creditType":"GOOGLE_ONE_AI","creditAmount":"42"}]}\n\n', - ].join(""); - - const transform = createCreditsExtractionTransform("test-account"); - const readable = new ReadableStream({ - start(controller) { - controller.enqueue(encoder.encode(sseData)); - controller.close(); - }, - }); - - // Consume the stream through the transform - const output = readable.pipeThrough(transform); - const reader = output.getReader(); - const chunks: Uint8Array[] = []; - while (true) { - const { done, value } = await reader.read(); - if (done) break; - chunks.push(value); - } - - // Data must pass through unmodified - const collected = new TextDecoder().decode( - new Uint8Array( - chunks.reduce((acc, c) => acc + c.length, 0) > 0 ? Buffer.concat(chunks) : new Uint8Array(0) - ) - ); - assert.ok(collected.includes("hello")); - assert.ok(collected.includes("remainingCredits")); -}); - -test("createCreditsExtractionTransform with buffer cap truncates large buffers", async () => { - const encoder = new TextEncoder(); - // Build a payload larger than 1KB - const largeText = "x".repeat(2000); - const ssePayload = JSON.stringify({ - response: { - candidates: [{ content: { parts: [{ text: largeText }] } }], - }, - }); - const sseLine = `data: ${ssePayload}\n\n`; - // Append a credits line at the end - const creditsLine = - 'data: {"remainingCredits":[{"creditType":"GOOGLE_ONE_AI","creditAmount":"99"}]}\n\n'; - const fullData = sseLine + creditsLine; - - // Use a 512-byte buffer cap -- the large text line should be discarded - const transform = createCreditsExtractionTransform("test-account", 512); - const readable = new ReadableStream({ - start(controller) { - // Send in small chunks to exercise the sliding-window logic - const encoded = encoder.encode(fullData); - const chunkSize = 256; - for (let i = 0; i < encoded.length; i += chunkSize) { - controller.enqueue(encoded.slice(i, i + chunkSize)); - } - controller.close(); - }, - }); - - const output = readable.pipeThrough(transform); - const reader = output.getReader(); - while (true) { - const { done } = await reader.read(); - if (done) break; - } - - // The transform should not throw -- buffer cap just limits what the - // flush handler can see. If the credits line was within the last 512 - // bytes it will be found; otherwise it's a graceful no-op. - // Either way, no crash or OOM. - assert.ok(true); -}); - -test("createCreditsExtractionTransform handles malformed SSE gracefully", async () => { - const encoder = new TextEncoder(); - const badData = "not valid sse\ndata: {broken json\n\ndata: [DONE]\n\n"; - - const transform = createCreditsExtractionTransform("test-account"); - const readable = new ReadableStream({ - start(controller) { - controller.enqueue(encoder.encode(badData)); - controller.close(); - }, - }); - - const output = readable.pipeThrough(transform); - const reader = output.getReader(); - const chunks: Uint8Array[] = []; - while (true) { - const { done, value } = await reader.read(); - if (done) break; - chunks.push(value); - } - - // Data passes through unmodified, no crash on malformed input - const collected = new TextDecoder().decode(Buffer.concat(chunks)); - assert.ok(collected.includes("not valid sse")); -}); diff --git a/tests/unit/sse-parser.test.ts b/tests/unit/sse-parser.test.ts index dc7568af5c8..36fcd7dedac 100644 --- a/tests/unit/sse-parser.test.ts +++ b/tests/unit/sse-parser.test.ts @@ -1,12 +1,10 @@ import test from "node:test"; import assert from "node:assert/strict"; -const { - parseSSEToOpenAIResponse, - parseSSEToClaudeResponse, - parseSSEToResponsesOutput, - parseSSEToGeminiResponse, -} = await import("../../open-sse/handlers/sseParser.ts"); +const { parseSSEToOpenAIResponse, parseSSEToClaudeResponse, parseSSEToResponsesOutput } = + await import("../../open-sse/handlers/sseParser.ts"); +const { parseSSEToGeminiResponse } = + await import("../../open-sse/handlers/sseParser/geminiResponse.ts"); test("parseSSEToOpenAIResponse parses a single SSE event with a done marker", () => { const rawSSE = [ From a9574d8e60210f345a6788dd6e30fb946354095d Mon Sep 17 00:00:00 2001 From: HouMinXi <19586012+HouMinXi@users.noreply.github.com> Date: Sat, 18 Jul 2026 14:24:16 -0300 Subject: [PATCH 12/14] refactor(antigravity): decompose execute + gemini parser below complexity gate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit executeOnce() (complexity 127, 436 lines) and parseSSEToGeminiResponse() (complexity 39, 117 lines) were both over the check-complexity.mjs gate (complexity>15, max-lines-per-function>80). Decomposed each into small named helpers, no behavior change: - geminiResponse.ts: split into pure per-concern functions (markdown shortcut, candidate-parts walk, finishReason, usageMetadata, final response assembly). - antigravity.ts: extracted the per-url-index attempt pipeline (runAntigravityAttempt, handleAntigravityRateLimit, tryResolveRetryFromErrorBody, shouldAutoRetryTransient) and moved the request/result-building helpers (send, credits-retry, embed-retry, non-streaming/streaming result builders) into a new antigravity/executeAttempt.ts submodule, mirroring the existing streamingPassthrough.ts/sseCollect.ts pattern. Also fixes the antigravity.ts file-size cap (was pushed to 2084 lines > 1813 frozen ceiling by the decomposition itself; now 1428). check-complexity.mjs: 2054 violations (baseline 2058) — net improvement. execute/executeOnce/parseSSEToGeminiResponse no longer appear with ruleId complexity or max-lines-per-function. Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> --- open-sse/executors/antigravity.ts | 961 +++++++----------- .../executors/antigravity/executeAttempt.ts | 687 +++++++++++++ open-sse/handlers/sseParser/geminiResponse.ts | 290 +++--- 3 files changed, 1209 insertions(+), 729 deletions(-) create mode 100644 open-sse/executors/antigravity/executeAttempt.ts diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index 283cd6689c0..e3fc66b378b 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -1,21 +1,16 @@ import crypto, { randomUUID } from "crypto"; import { BaseExecutor, - mergeAbortSignals, mergeUpstreamExtraHeaders, type ExecuteInput, type ExecutorLog, type ProviderCredentials, } from "./base.ts"; -import { applyFingerprint, isCliCompatEnabled } from "../config/cliFingerprints.ts"; -import { buildAntigravityUpstreamError } from "./antigravityUpstreamError.ts"; import { PROVIDERS, OAUTH_ENDPOINTS, HTTP_STATUS, FETCH_TIMEOUT_MS, - STREAM_READINESS_TIMEOUT_MS, - ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE, } from "../config/constants.ts"; import { scrubProxyAndFingerprintHeaders } from "../services/antigravityHeaderScrub.ts"; import { @@ -24,7 +19,6 @@ import { } from "../services/antigravityHeaders.ts"; import { classify429, decide429, type Decision } from "../services/antigravity429Engine.ts"; import { - injectCreditsField, shouldRetryWithCredits, shouldUseCreditsFirst, getCreditsMode, @@ -40,7 +34,6 @@ import { resolveAntigravityModelId, getAntigravityModelFallbacks, } from "../config/antigravityModelAliases.ts"; -import { cloakAntigravityToolPayload } from "../config/toolCloaking.ts"; import { shouldStripCloudCodeThinking, stripCloudCodeThinkingConfig, @@ -56,25 +49,33 @@ import { export { processAntigravitySSEPayload } from "./antigravity/sseCollect.ts"; import { createCreditsExtractionTransform as createCreditsExtractionTransformImpl, - buildSsePassthroughResult, + type SsePassthroughResult, } from "./antigravity/streamingPassthrough.ts"; import { - applyAntigravityClientProfileHeaders, - removeHeaderCaseInsensitive, -} from "../services/antigravityClientProfile.ts"; + toSafeAntigravityLog, + finalizeAntigravityRequestBody, + sendAntigravityRequest, + tryCreditsRetry, + tryEmbedLongRetryAfter, + buildFinalAntigravityResult, + buildAntigravity429ErrorMessage, + markCreditsExhausted, + type SafeAntigravityLog, +} from "./antigravity/executeAttempt.ts"; import { generateAntigravityRequestId, getAntigravityEnvelopeUserAgent, getAntigravitySessionId, } from "../services/antigravityIdentity.ts"; -import * as prl from "../utils/providerRequestLogging.ts"; const MAX_RETRY_AFTER_MS = 60_000; const LONG_RETRY_THRESHOLD_MS = 60_000; -const CREDITS_EXHAUSTED_TTL_MS = 5 * 60 * 60 * 1000; // 5 hours // Cap for transient 5xx backoff — shorter than the 429 cap to avoid long stalls on // infra hiccups ("Agent execution terminated", "high traffic", capacity errors). const ANTIGRAVITY_TRANSIENT_RETRY_MAX_MS = 15_000; +// Bounded per-URL auto-retry count for both the Retry-After-driven short retry and +// the no-Retry-After transient/429 backoff loop in executeOnce(). +const MAX_AUTO_RETRIES = 3; const ANTIGRAVITY_TRANSIENT_ERROR_PATTERNS: RegExp[] = [ /high\s+traffic/i, @@ -106,7 +107,7 @@ interface AntigravityContent { [key: string]: unknown; } -type AntigravityCredentials = ProviderCredentials & { +export type AntigravityCredentials = ProviderCredentials & { projectId?: string | null; expiresIn?: number; }; @@ -124,46 +125,6 @@ type AntigravityChunkContent = Record & { >; }; -function getChunkedOrFixedBody(bodyStr: string, stream: boolean): BodyInit { - if (stream) { - return new ReadableStream( - { - async start(controller) { - controller.enqueue(new TextEncoder().encode(bodyStr)); - controller.close(); - }, - }, - { highWaterMark: 16384 } - ); - } - return bodyStr; -} - -function cloneAntigravityRequestBody(body: unknown): unknown { - if (!body || typeof body !== "object") { - return body; - } - - try { - return structuredClone(body); - } catch { - return JSON.parse(JSON.stringify(body)); - } -} - -function serializeAntigravityRequest( - provider: string, - headers: Record, - body: unknown -): { headers: Record; bodyString: string } { - const serializedBody = cloneAntigravityRequestBody(body); - - if (!isCliCompatEnabled(provider)) { - return { headers, bodyString: JSON.stringify(serializedBody) }; - } - return applyFingerprint(provider, { ...headers }, serializedBody); -} - type AntigravityRequestEnvelope = Record & { project: string; model?: string; @@ -174,44 +135,6 @@ type AntigravityRequestEnvelope = Record & { enabledCreditTypes?: string[]; }; -class AntigravityPreResponseTimeoutError extends Error { - code = ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE; - status = HTTP_STATUS.GATEWAY_TIMEOUT; - - constructor(timeoutMs: number, url: string) { - super(`Antigravity upstream did not return response headers within ${timeoutMs}ms: ${url}`); - this.name = "TimeoutError"; - } -} - -function getAbortErrorCode(error: unknown): string | null { - if (!error || typeof error !== "object") return null; - const value = (error as { code?: unknown }).code; - return typeof value === "string" ? value : null; -} - -function isAntigravityPreResponseTimeout(error: unknown): boolean { - return getAbortErrorCode(error) === ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE; -} - -/** - * Per-account GOOGLE_ONE_AI credits-exhausted tracker. - * Key: accountId (OAuth subject / email). Value: expiry timestamp. - * When credits hit 0 we skip the credit retry for CREDITS_EXHAUSTED_TTL_MS. - */ -const MAX_CREDITS_EXHAUSTED_ENTRIES = 50; -const creditsExhaustedUntil = new Map(); - -const _creditsExhaustedSweep = setInterval(() => { - const now = Date.now(); - for (const [key, until] of creditsExhaustedUntil) { - if (now >= until) creditsExhaustedUntil.delete(key); - } -}, 60_000); -if (typeof _creditsExhaustedSweep === "object" && "unref" in _creditsExhaustedSweep) { - (_creditsExhaustedSweep as { unref?: () => void }).unref?.(); -} - const MAX_CREDIT_BALANCE_ENTRIES = 50; const CREDIT_BALANCE_TTL_MS = 5 * 60 * 1000; const creditBalanceCache = new Map(); @@ -291,35 +214,6 @@ export function createCreditsExtractionTransform( ); } -function isCreditsExhausted(accountId: string): boolean { - const until = creditsExhaustedUntil.get(accountId); - if (!until) return false; - if (Date.now() >= until) { - creditsExhaustedUntil.delete(accountId); - return false; - } - return true; -} - -function markCreditsExhausted(accountId: string): void { - if ( - creditsExhaustedUntil.size >= MAX_CREDITS_EXHAUSTED_ENTRIES && - !creditsExhaustedUntil.has(accountId) - ) { - const now = Date.now(); - for (const [key, until] of creditsExhaustedUntil) { - if (now >= until) { - creditsExhaustedUntil.delete(key); - } - } - if (creditsExhaustedUntil.size >= MAX_CREDITS_EXHAUSTED_ENTRIES) { - const oldestKey = creditsExhaustedUntil.keys().next().value; - if (oldestKey !== undefined) creditsExhaustedUntil.delete(oldestKey); - } - } - creditsExhaustedUntil.set(accountId, Date.now() + CREDITS_EXHAUSTED_TTL_MS); -} - /** * Persist a quota-exhausted cooldown to the DB for `connectionId` so that * cross-request and post-restart routing skips this connection until the @@ -382,26 +276,6 @@ async function cleanModelName(model: string, modelIdOverride?: string): Promise< return clean; } -function attachToolNameMap(payload: T, toolNameMap: Map | null): T { - if (!toolNameMap?.size || !payload || typeof payload !== "object") { - return payload; - } - - const copy = Array.isArray(payload) ? ([...payload] as T) : ({ ...(payload as object) } as T); - Object.defineProperty(copy, "_toolNameMap", { - value: toolNameMap, - enumerable: false, - configurable: true, - writable: true, - }); - return copy; -} - -function getRequestTargetModel(body: Record): string { - const target = body.model; - return typeof target === "string" && target.length > 0 ? target : "unknown"; -} - /** * Hard ceiling on `generationConfig.maxOutputTokens` for Antigravity Cloud Code. * @@ -551,6 +425,49 @@ function stripTrailingAntigravityAssistantTurn( // Test-only export so the unit suite can exercise the strip logic directly. export const __test_stripTrailingAntigravityAssistantTurn = stripTrailingAntigravityAssistantTurn; +/** Base per-url-index attempt context, before the request has been sent. */ +type AntigravityAttemptContext = { + url: string; + model: string; + /** Pre-serialization headers (built by buildHeaders + mergeUpstreamExtraHeaders) — the + * credits-retry re-serializes from these, NOT from `finalHeaders` (already fingerprinted). */ + headers: Record; + transformedBody: Record; + requestToolNameMap: Map | null; + credentials: AntigravityCredentials; + stream: boolean; + signal: AbortSignal | null | undefined; + log: SafeAntigravityLog; + accountId: string; + creditsMode: ReturnType; + urlIndex: number; + retryAttemptsByUrl: Record; + fallbackCount: number; +}; + +/** Context threaded through the 429/503 handling helpers — adds the sent response. */ +type AntigravityRateLimitContext = AntigravityAttemptContext & { + response: Response; + finalHeaders: Record; +}; + +/** + * Outcome of handling a 429/503 response — tells executeOnce()'s loop what to do next. + * `lastStatus` mirrors the original inline code, which only updated the outer + * `lastStatus` variable when NOT retrying the same url (i.e. on retryNextUrl/fallthrough, + * never on the bounded-short-retry or transient-auto-retry same-url paths). + */ +type AntigravityRateLimitOutcome = + | { action: "return"; result: SsePassthroughResult } + | { action: "retrySameUrl" } + | { action: "retryNextUrl"; lastStatus: number } + | { action: "fallthrough"; retryMs: number | null; lastStatus: number }; + +/** Outcome of one full per-url attempt in executeOnce() — return a result, or retry. */ +type AntigravityAttemptOutcome = + | { action: "return"; result: SsePassthroughResult } + | { action: "retry"; sameUrl: boolean; lastStatus?: number }; + export class AntigravityExecutor extends BaseExecutor { constructor() { super("antigravity", PROVIDERS.antigravity); @@ -1142,9 +1059,9 @@ export class AntigravityExecutor extends BaseExecutor { ) { await resolveAntigravityVersion(); const fallbackCount = this.getFallbackCount(); + const l = toSafeAntigravityLog(log); let lastError = null; let lastStatus = 0; - const MAX_AUTO_RETRIES = 3; const retryAttemptsByUrl: Record = {}; // Track retry attempts per URL // Always stream upstream — buildUrl always returns the streaming endpoint. @@ -1164,44 +1081,6 @@ export class AntigravityExecutor extends BaseExecutor { const creditsMode = getCreditsMode(); const useCreditsFirst = shouldUseCreditsFirst(credentials?.accessToken || "", creditsMode); - const fetchWithReadinessTimeout = async ( - url: string, - init: RequestInit, - timeoutMs = STREAM_READINESS_TIMEOUT_MS - ): Promise => { - const boundedTimeoutMs = Math.max(0, Math.floor(timeoutMs)); - if (boundedTimeoutMs <= 0) { - return fetch(url, init); - } - - const timeoutController = new AbortController(); - let timeoutId: ReturnType | null = setTimeout(() => { - timeoutController.abort(new AntigravityPreResponseTimeoutError(boundedTimeoutMs, url)); - }, boundedTimeoutMs); - - const existingSignal = init.signal instanceof AbortSignal ? init.signal : null; - const combinedSignal = existingSignal - ? mergeAbortSignals(existingSignal, timeoutController.signal) - : timeoutController.signal; - - try { - return await fetch(url, { ...init, signal: combinedSignal }); - } catch (error) { - if ( - timeoutController.signal.aborted && - isAntigravityPreResponseTimeout(timeoutController.signal.reason) - ) { - throw timeoutController.signal.reason; - } - throw error; - } finally { - if (timeoutId) { - clearTimeout(timeoutId); - timeoutId = null; - } - } - }; - for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) { const url = this.buildUrl(model, upstreamStream, urlIndex); const headers = this.buildHeaders(credentials, upstreamStream); @@ -1213,27 +1092,16 @@ export class AntigravityExecutor extends BaseExecutor { credentials, modelIdOverride ); - let requestToolNameMap: Map | null = null; if (transformed instanceof Response) { return { response: transformed, url, headers, transformedBody: body }; } - let transformedBody: Record = transformed; - - if (transformedBody && typeof transformedBody === "object") { - const cloaked = cloakAntigravityToolPayload(transformedBody); - transformedBody = cloaked.body; - requestToolNameMap = cloaked.toolNameMap; - } - - // Credits-first: inject GOOGLE_ONE_AI upfront so we never try the normal - // quota path. If credits are exhausted / disabled shouldUseCreditsFirst() - // returns false and we fall back to the legacy retry-on-429 flow. - if (useCreditsFirst) { - transformedBody = injectCreditsField(transformedBody); - log?.debug?.("AG_CREDITS", "Credits-first enabled (ANTIGRAVITY_CREDITS=always)"); - } + const { transformedBody, requestToolNameMap } = finalizeAntigravityRequestBody( + transformed, + useCreditsFirst, + l + ); // Initialize retry counter for this URL if (!retryAttemptsByUrl[urlIndex]) { @@ -1241,452 +1109,319 @@ export class AntigravityExecutor extends BaseExecutor { } try { - const serializedRequest = serializeAntigravityRequest( - this.provider, + const outcome = await this.runAntigravityAttempt({ + url, + model, headers, - transformedBody - ); - let finalHeaders = serializedRequest.headers; - const capture = (h: Record, s: string) => - prl.captureCurrentProviderBody(url, h, s, log); - const clientProfile = applyAntigravityClientProfileHeaders( - finalHeaders, + transformedBody, + requestToolNameMap, credentials, - transformedBody - ); + stream, + signal, + log: l, + accountId, + creditsMode, + urlIndex, + retryAttemptsByUrl, + fallbackCount, + }); - log?.debug?.( + if (outcome.action === "return") return outcome.result; + if (outcome.lastStatus !== undefined) lastStatus = outcome.lastStatus; + if (outcome.sameUrl) urlIndex--; + continue; + } catch (error) { + lastError = error; + l.error( "TELEMETRY", - `[Antigravity] Execute - URL: ${url}, Model: ${model}, Target: ${getRequestTargetModel(transformedBody)}, RetryAttempt: ${retryAttemptsByUrl[urlIndex]}` + `[Antigravity] Network/Fetch Error - URL: ${url}, Model: ${model}, Error: ${error instanceof Error ? error.message : String(error)}` ); - - // Dump outgoing headers (mask Authorization) and envelope shape for debugging - if (log?.debug) { - const safeHeaders = { ...finalHeaders }; - if (safeHeaders["Authorization"]) safeHeaders["Authorization"] = "Bearer ***"; - log.debug("AG_REQUEST_HEADERS", JSON.stringify(safeHeaders)); - - const envelope = transformedBody as Record; - const requestInner = envelope.request as Record | undefined; - log.debug( - "AG_REQUEST_ENVELOPE", - JSON.stringify({ - fieldOrder: Object.keys(envelope), - project: envelope.project, - requestId: envelope.requestId, - model: envelope.model, - userAgent: envelope.userAgent, - requestType: envelope.requestType, - enabledCreditTypes: envelope.enabledCreditTypes, - clientProfile, - sessionId: requestInner?.sessionId, - generationConfig: requestInner?.generationConfig, - }) - ); + if (urlIndex + 1 < fallbackCount) { + l.debug("RETRY", `Error on ${url}, trying fallback ${urlIndex + 1}`); + continue; } + throw error; + } + } - await capture(finalHeaders, serializedRequest.bodyString); - let response = await fetchWithReadinessTimeout(url, { - method: "POST", - headers: finalHeaders, - body: getChunkedOrFixedBody(serializedRequest.bodyString, stream), - ...(stream ? { duplex: "half" } : {}), - signal, - }); + throw lastError || new Error(`All ${fallbackCount} URLs failed with status ${lastStatus}`); + } - if (response.status === HTTP_STATUS.FORBIDDEN && finalHeaders["x-goog-user-project"]) { - const retryHeaders = { ...finalHeaders }; - removeHeaderCaseInsensitive(retryHeaders, "x-goog-user-project"); - log?.debug?.("RETRY", "403 with x-goog-user-project, retrying once without it"); - await capture(retryHeaders, serializedRequest.bodyString); - response = await fetchWithReadinessTimeout(url, { - method: "POST", - headers: retryHeaders, - body: getChunkedOrFixedBody(serializedRequest.bodyString, stream), - ...(stream ? { duplex: "half" } : {}), - signal, - }); - finalHeaders = retryHeaders; - } + /** + * Run one full per-url-index attempt: send the request, handle a 429/503 (retry + * same/next url, or a Google One AI credits retry), fall back on other retryable + * statuses, optionally embed a long Retry-After, then build the final non-streaming + * or streaming result. Returns a result to hand back from execute(), or a retry + * instruction for executeOnce()'s loop to act on (continue, optionally urlIndex--). + */ + private async runAntigravityAttempt( + ctx: AntigravityAttemptContext + ): Promise { + const { + url, + model, + headers, + transformedBody, + requestToolNameMap, + credentials, + stream, + signal, + log, + accountId, + urlIndex, + retryAttemptsByUrl, + fallbackCount, + } = ctx; + + const { response, finalHeaders } = await sendAntigravityRequest( + this.provider, + url, + model, + headers, + transformedBody, + credentials, + stream, + signal, + log, + retryAttemptsByUrl[urlIndex] + ); - if (!response.ok) { - log?.warn?.( - "TELEMETRY", - `[Antigravity] Error Response - URL: ${url}, Status: ${response.status}, Model: ${model}` - ); - } + let retryMs: number | null = null; - // Parse retry time for 429/503 responses - let retryMs: number | null = null; - - if ( - response.status === HTTP_STATUS.RATE_LIMITED || - response.status === HTTP_STATUS.SERVICE_UNAVAILABLE - ) { - // Try to get retry time from headers first - retryMs = this.parseRetryHeaders(response.headers); - - // If no retry time in headers, try to parse from error message body - if (!retryMs) { - try { - const errorBody = await response.clone().text(); - const errorJson = JSON.parse(errorBody); - let errorMessage = errorJson?.error?.message || errorJson?.message || ""; - if (errorJson?.error?.details && Array.isArray(errorJson.error.details)) { - for (const detail of errorJson.error.details) { - if (detail?.reason) { - errorMessage += ` ${detail.reason}`; - } - } - } - - // 1. Try to parse explicit retry time from message - const parsedRetryMs = this.parseRetryFromErrorMessage(errorMessage); - - // 2. Classify 429 (pass header-parsed retry hint as fallback - // signal — multi-hour Retry-After upgrades rate_limited to - // quota_exhausted so the GOOGLE_ONE_AI credits retry fires). - const effectiveRetryHintMs = retryMs ?? parsedRetryMs ?? null; - const category = classify429(errorMessage); - - // 3. Decide final retry time BEFORE the credits retry so that - // full_quota_exhausted can skip the credits attempt entirely - // (avoids ~41s hold on an already-exhausted account) and - // persist the cooldown to DB for post-restart routing. - const decision: Decision = decide429(category, parsedRetryMs); - retryMs = decision.retryAfterMs; - log?.debug?.( - "AG_429", - `Category: ${category}, Decision: ${decision.kind} — ${decision.reason}` - ); - - if (decision.kind === "full_quota_exhausted" && retryMs) { - markConnectionQuotaExhausted(accountId, retryMs); - } - - const creditsAlreadyInjected = - (transformedBody as { enabledCreditTypes?: unknown }).enabledCreditTypes != null; - - if (category === "quota_exhausted" && creditsAlreadyInjected) { - handleCreditsFailure(credentials?.accessToken || ""); - log?.warn?.("AG_CREDITS", "Credits-first request 429'd — credits likely exhausted"); - markCreditsExhausted(accountId); - } - - if ( - category === "quota_exhausted" && - decision.kind !== "full_quota_exhausted" && - !creditsAlreadyInjected && - shouldRetryWithCredits(credentials?.accessToken || "", creditsMode !== "off") - ) { - log?.info?.("AG_CREDITS", "Retrying with Google One AI credits"); - const creditsBody = injectCreditsField(transformedBody); - const serializedCreditsRequest = serializeAntigravityRequest( - this.provider, - headers, - creditsBody - ); - const finalCreditsHeaders = serializedCreditsRequest.headers; - try { - await capture(finalCreditsHeaders, serializedCreditsRequest.bodyString); - const creditsResp = await fetchWithReadinessTimeout(url, { - method: "POST", - headers: finalCreditsHeaders, - body: getChunkedOrFixedBody(serializedCreditsRequest.bodyString, stream), - ...(stream ? { duplex: "half" } : {}), - signal, - }); - if (creditsResp.ok || creditsResp.status !== HTTP_STATUS.RATE_LIMITED) { - log?.info?.("AG_CREDITS", `Credits retry succeeded: ${creditsResp.status}`); - if (!stream && creditsResp.body) { - // Raw SSE pass-through + credits extraction (see - // streamingPassthrough.ts); 499s early if the client - // already disconnected instead of piping a cancelled body. - return buildSsePassthroughResult( - creditsResp.body, - creditsResp, - accountId, - updateAntigravityRemainingCredits, - url, - finalCreditsHeaders, - attachToolNameMap(creditsBody, requestToolNameMap), - signal - ); - } - return { - response: creditsResp, - url, - headers: finalCreditsHeaders, - transformedBody: attachToolNameMap(creditsBody, requestToolNameMap), - }; - } + if ( + response.status === HTTP_STATUS.RATE_LIMITED || + response.status === HTTP_STATUS.SERVICE_UNAVAILABLE + ) { + const rateLimitOutcome = await this.handleAntigravityRateLimit({ + ...ctx, + response, + finalHeaders, + }); - // Credit retry also 429'd - handleCreditsFailure(credentials?.accessToken || ""); - log?.warn?.("AG_CREDITS", "Credits retry also 429'd"); - - // Also mark in our legacy exhaustion map to avoid retrying other routes - markCreditsExhausted(accountId); - } catch (creditsErr) { - handleCreditsFailure(credentials?.accessToken || ""); - log?.warn?.("AG_CREDITS", `Credits retry failed: ${creditsErr}`); - } - } - } catch (e) { - // Ignore parse errors, will fall back to exponential backoff - } - } + if (rateLimitOutcome.action === "return") { + return { action: "return", result: rateLimitOutcome.result }; + } + if (rateLimitOutcome.action === "retrySameUrl") return { action: "retry", sameUrl: true }; + if (rateLimitOutcome.action === "retryNextUrl") { + return { action: "retry", sameUrl: false, lastStatus: rateLimitOutcome.lastStatus }; + } + // Only "fallthrough" remains: last url, no more retries — proceed below with + // the resolved retryMs so a long Retry-After can still be embedded in the body. + retryMs = rateLimitOutcome.retryMs; + } - // Bounded short-retry: a non-null retryAfterMs ≤ 60s covers nearly every - // 429 (decide429 returns 2s/5s/60s defaults), so this branch MUST share the - // per-URL attempt counter. Without the bound a persistent 429 loops forever - // on the same endpoint/account (urlIndex-- cancels the loop's urlIndex++) and - // never returns the 429 to the account-fallback layer in chat.ts. - if ( - retryMs && - retryMs <= LONG_RETRY_THRESHOLD_MS && - retryAttemptsByUrl[urlIndex] < MAX_AUTO_RETRIES - ) { - retryAttemptsByUrl[urlIndex]++; - const effectiveRetryMs = Math.min(retryMs, MAX_RETRY_AFTER_MS); - log?.debug?.( - "RETRY", - `${response.status} retry ${retryAttemptsByUrl[urlIndex]}/${MAX_AUTO_RETRIES} with Retry-After: ${Math.ceil(effectiveRetryMs / 1000)}s, waiting...` - ); - await new Promise((resolve) => setTimeout(resolve, effectiveRetryMs)); - urlIndex--; - continue; - } + if (this.shouldRetry(response.status, urlIndex)) { + log.debug("RETRY", `${response.status} on ${url}, trying fallback ${urlIndex + 1}`); + return { action: "retry", sameUrl: false, lastStatus: response.status }; + } - // Auto retry for 429 (no Retry-After) or transient 5xx errors. - // For 5xx we read the body to detect known transient patterns - // ("Agent execution terminated due to error", "high traffic", "capacity"). - if ((!retryMs || retryMs === 0) && retryAttemptsByUrl[urlIndex] < MAX_AUTO_RETRIES) { - let shouldAutoRetry = response.status === HTTP_STATUS.RATE_LIMITED; - if (!shouldAutoRetry && ANTIGRAVITY_TRANSIENT_STATUSES.has(response.status)) { - try { - const errBody = await response.clone().text(); - let errJson: unknown = null; - try { - errJson = errBody ? JSON.parse(errBody) : null; - } catch { - // non-JSON body — fall through to pattern match against raw text - } - const errMsg = this.extractErrorMessage(errJson, errBody); - shouldAutoRetry = this.isTransientAntigravityError(response.status, errMsg); - } catch { - // ignore body read errors - } - } - if (shouldAutoRetry) { - retryAttemptsByUrl[urlIndex]++; - // Exponential backoff: 2s, 4s, 8s… capped per-status - const cap = - response.status === HTTP_STATUS.RATE_LIMITED - ? MAX_RETRY_AFTER_MS - : ANTIGRAVITY_TRANSIENT_RETRY_MAX_MS; - const backoffMs = Math.min(1000 * 2 ** retryAttemptsByUrl[urlIndex], cap); - log?.debug?.( - "RETRY", - `${response.status} transient auto retry ${retryAttemptsByUrl[urlIndex]}/${MAX_AUTO_RETRIES} after ${backoffMs / 1000}s` - ); - await new Promise((resolve) => setTimeout(resolve, backoffMs)); - urlIndex--; - continue; - } - } + // If we have a 429 with long retry time, embed it in response body + const embedded = await tryEmbedLongRetryAfter( + response, + retryMs, + url, + finalHeaders, + transformedBody, + requestToolNameMap, + log + ); + if (embedded) return { action: "return", result: embedded }; + + const result = await buildFinalAntigravityResult( + stream, + response, + url, + finalHeaders, + transformedBody, + requestToolNameMap, + accountId, + signal, + updateAntigravityRemainingCredits + ); + return { action: "return", result }; + } - log?.debug?.( - "RETRY", - `${response.status}, Retry-After ${retryMs ? `too long (${Math.ceil(retryMs / 1000)}s)` : "missing"}, trying fallback` - ); - lastStatus = response.status; + /** + * Handle a 429/503 response for one URL-index attempt: resolve the retry-after + * time (headers, then error-body classification + Google-One-AI credits retry), + * then decide whether to retry the SAME url, fall back to the NEXT url, or (on + * the last url with no more retries left) fall through with the resolved retryMs + * so the caller can still embed a long Retry-After in the final response body. + */ + private async handleAntigravityRateLimit( + ctx: AntigravityRateLimitContext + ): Promise { + const { response, log, urlIndex, retryAttemptsByUrl, fallbackCount } = ctx; + + // Try to get retry time from headers first + let retryMs: number | null = this.parseRetryHeaders(response.headers); + + // If no retry time in headers, try to parse from error message body + if (!retryMs) { + const resolved = await this.tryResolveRetryFromErrorBody(ctx); + if (resolved.kind === "return") return { action: "return", result: resolved.result }; + retryMs = resolved.retryMs; + } - if (urlIndex + 1 < fallbackCount) { - continue; - } - } + // Bounded short-retry: a non-null retryAfterMs ≤ 60s covers nearly every + // 429 (decide429 returns 2s/5s/60s defaults), so this branch MUST share the + // per-URL attempt counter. Without the bound a persistent 429 loops forever + // on the same endpoint/account (urlIndex-- cancels the loop's urlIndex++) and + // never returns the 429 to the account-fallback layer in chat.ts. + if ( + retryMs && + retryMs <= LONG_RETRY_THRESHOLD_MS && + retryAttemptsByUrl[urlIndex] < MAX_AUTO_RETRIES + ) { + retryAttemptsByUrl[urlIndex]++; + const effectiveRetryMs = Math.min(retryMs, MAX_RETRY_AFTER_MS); + log.debug( + "RETRY", + `${response.status} retry ${retryAttemptsByUrl[urlIndex]}/${MAX_AUTO_RETRIES} with Retry-After: ${Math.ceil(effectiveRetryMs / 1000)}s, waiting...` + ); + await new Promise((resolve) => setTimeout(resolve, effectiveRetryMs)); + return { action: "retrySameUrl" }; + } - if (this.shouldRetry(response.status, urlIndex)) { - log?.debug?.("RETRY", `${response.status} on ${url}, trying fallback ${urlIndex + 1}`); - lastStatus = response.status; - continue; - } + // Auto retry for 429 (no Retry-After) or transient 5xx errors. + // For 5xx we read the body to detect known transient patterns + // ("Agent execution terminated due to error", "high traffic", "capacity"). + if ((!retryMs || retryMs === 0) && retryAttemptsByUrl[urlIndex] < MAX_AUTO_RETRIES) { + const shouldAutoRetry = await this.shouldAutoRetryTransient(response); + if (shouldAutoRetry) { + retryAttemptsByUrl[urlIndex]++; + // Exponential backoff: 2s, 4s, 8s… capped per-status + const cap = + response.status === HTTP_STATUS.RATE_LIMITED + ? MAX_RETRY_AFTER_MS + : ANTIGRAVITY_TRANSIENT_RETRY_MAX_MS; + const backoffMs = Math.min(1000 * 2 ** retryAttemptsByUrl[urlIndex], cap); + log.debug( + "RETRY", + `${response.status} transient auto retry ${retryAttemptsByUrl[urlIndex]}/${MAX_AUTO_RETRIES} after ${backoffMs / 1000}s` + ); + await new Promise((resolve) => setTimeout(resolve, backoffMs)); + return { action: "retrySameUrl" }; + } + } - // If we have a 429 with long retry time, embed it in response body - if ( - response.status === HTTP_STATUS.RATE_LIMITED && - retryMs && - retryMs > LONG_RETRY_THRESHOLD_MS - ) { - try { - const respBody = await response.clone().text(); - let obj; - try { - obj = JSON.parse(respBody); - } catch { - obj = {}; - } - obj.retryAfterMs = retryMs; - const modifiedBody = JSON.stringify(obj); - const modifiedResponse = new Response(modifiedBody, { - status: response.status, - headers: response.headers, - }); - return { - response: modifiedResponse, - url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; - } catch (err) { - log?.warn?.("RETRY", `Failed to embed retryAfterMs: ${err}`); - // Fall back to original response - } - } + log.debug( + "RETRY", + `${response.status}, Retry-After ${retryMs ? `too long (${Math.ceil(retryMs / 1000)}s)` : "missing"}, trying fallback` + ); - // For non-streaming clients, return the raw SSE stream with a - // credits-extraction TransformStream. chatCore's non-streaming path - // (readNonStreamingResponseBody + parseNonStreamingSSEPayload with - // Gemini format support) handles draining and conversion to JSON. - // This replaces the previous collectStreamToResponse() approach which - // had an artificial timeout (now the standard FETCH_BODY_TIMEOUT_MS - // of 10 min applies). - if (!stream) { - // #3229: surface a real upstream error instead of masking a 4xx/5xx as an - // empty `chat.completion` envelope. - if (!response.ok) { - const rawBody = await response - .clone() - .text() - .catch(() => ""); - const errorBody = buildAntigravityUpstreamError( - response.status, - response.statusText, - rawBody - ); - return { - response: new Response(JSON.stringify(errorBody), { - status: response.status, - headers: { "Content-Type": "application/json" }, - }), - url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; - } + if (urlIndex + 1 < fallbackCount) { + return { action: "retryNextUrl", lastStatus: response.status }; + } - if (response.body) { - // Raw SSE pass-through + credits extraction (see - // streamingPassthrough.ts); 499s early if the client already - // disconnected instead of piping a cancelled body. - return buildSsePassthroughResult( - response.body, - response, - accountId, - updateAntigravityRemainingCredits, - url, - finalHeaders, - attachToolNameMap(transformedBody, requestToolNameMap), - signal - ); - } + return { action: "fallthrough", retryMs, lastStatus: response.status }; + } - // No body -- return as-is - return { - response, - url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; - } + /** + * Parse the 429/503 response body to classify the failure and (for + * quota_exhausted, non-full-exhaustion cases) attempt a Google One AI + * credits retry. Returns the resolved retryMs, or an early "return" result + * when the credits retry itself produced a response to hand back to the client. + */ + private async tryResolveRetryFromErrorBody( + ctx: AntigravityRateLimitContext + ): Promise< + { kind: "return"; result: SsePassthroughResult } | { kind: "resolved"; retryMs: number | null } + > { + const { + response, + url, + headers, + transformedBody, + requestToolNameMap, + credentials, + stream, + signal, + log, + accountId, + creditsMode, + } = ctx; - // #2461: a non-ok upstream response (e.g. 403) must never be piped through the - // streaming pass-through below as if it were an SSE body. Google occasionally - // returns non-UTF8/binary error bodies (observed: gzip-magic-byte payloads) for - // 403s on this endpoint; reading/forwarding those raw bytes corrupts the - // client-visible error message. Mirror the non-streaming branch above and build - // a sanitized JSON error via buildAntigravityUpstreamError (hard rule #12) - // instead of streaming unknown bytes straight through. - if (!response.ok) { - const rawBody = await response - .clone() - .text() - .catch(() => ""); - const errorBody = buildAntigravityUpstreamError( - response.status, - response.statusText, - rawBody - ); - return { - response: new Response(JSON.stringify(errorBody), { - status: response.status, - headers: { "Content-Type": "application/json" }, - }), - url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; - } + try { + const errorBody = await response.clone().text(); + const errorJson = JSON.parse(errorBody); + const errorMessage = buildAntigravity429ErrorMessage(errorJson); + + // 1. Try to parse explicit retry time from message + const parsedRetryMs = this.parseRetryFromErrorMessage(errorMessage); + + // 2. Classify 429, then decide the final retry time BEFORE the credits + // retry so that full_quota_exhausted can skip the credits attempt + // entirely (avoids ~41s hold on an already-exhausted account) and + // persist the cooldown to DB for post-restart routing. + const category = classify429(errorMessage); + const decision: Decision = decide429(category, parsedRetryMs); + const retryMs = decision.retryAfterMs; + log.debug("AG_429", `Category: ${category}, Decision: ${decision.kind} — ${decision.reason}`); + + if (decision.kind === "full_quota_exhausted" && retryMs) { + markConnectionQuotaExhausted(accountId, retryMs); + } - // Streaming path: wrap the response body in a pass-through TransformStream - // that extracts remainingCredits from the final SSE chunk(s) without - // consuming the stream. The client receives the unmodified SSE data. - if (response.body) { - // If the downstream client aborts, cancel the upstream fetch body immediately - // to release the socket back to the Undici agent pool and prevent memory leaks. - if (signal) { - const abortHandler = () => { - try { - response.body?.cancel().catch(() => {}); - } catch (_) {} - }; - if (signal.aborted) { - abortHandler(); - } else { - signal.addEventListener("abort", abortHandler, { once: true }); - } - } + const creditsAlreadyInjected = + (transformedBody as { enabledCreditTypes?: unknown }).enabledCreditTypes != null; - const passThrough = createCreditsExtractionTransform( - accountId, - 16 * 1024 // 16KB sliding-window cap to prevent OOM - ); - const tappedBody = response.body.pipeThrough(passThrough); - const tappedResponse = new Response(tappedBody, { - status: response.status, - statusText: response.statusText, - headers: response.headers, - }); - return { - response: tappedResponse, - url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; - } + if (category === "quota_exhausted" && creditsAlreadyInjected) { + handleCreditsFailure(credentials?.accessToken || ""); + log.warn("AG_CREDITS", "Credits-first request 429'd — credits likely exhausted"); + markCreditsExhausted(accountId); + } - return { - response, + if ( + category === "quota_exhausted" && + decision.kind !== "full_quota_exhausted" && + !creditsAlreadyInjected && + shouldRetryWithCredits(credentials?.accessToken || "", creditsMode !== "off") + ) { + const creditsResult = await tryCreditsRetry( + this.provider, url, - headers: finalHeaders, - transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), - }; - } catch (error) { - lastError = error; - log?.error?.( - "TELEMETRY", - `[Antigravity] Network/Fetch Error - URL: ${url}, Model: ${model}, Error: ${error instanceof Error ? error.message : String(error)}` + headers, + transformedBody, + requestToolNameMap, + credentials, + stream, + signal, + log, + accountId, + updateAntigravityRemainingCredits ); - if (urlIndex + 1 < fallbackCount) { - log?.debug?.("RETRY", `Error on ${url}, trying fallback ${urlIndex + 1}`); - continue; - } - throw error; + if (creditsResult) return { kind: "return", result: creditsResult }; } + + return { kind: "resolved", retryMs }; + } catch { + // Ignore parse errors, will fall back to exponential backoff + return { kind: "resolved", retryMs: null }; } + } - throw lastError || new Error(`All ${fallbackCount} URLs failed with status ${lastStatus}`); + /** + * True for 429 always; for transient 5xx (500/502/503/504) only when the body + * matches a known capacity/traffic/agent-terminated pattern. + */ + private async shouldAutoRetryTransient(response: Response): Promise { + if (response.status === HTTP_STATUS.RATE_LIMITED) return true; + if (!ANTIGRAVITY_TRANSIENT_STATUSES.has(response.status)) return false; + try { + const errBody = await response.clone().text(); + let errJson: unknown = null; + try { + errJson = errBody ? JSON.parse(errBody) : null; + } catch { + // non-JSON body — fall through to pattern match against raw text + } + const errMsg = this.extractErrorMessage(errJson, errBody); + return this.isTransientAntigravityError(response.status, errMsg); + } catch { + // ignore body read errors + return false; + } } } diff --git a/open-sse/executors/antigravity/executeAttempt.ts b/open-sse/executors/antigravity/executeAttempt.ts new file mode 100644 index 00000000000..566e6517132 --- /dev/null +++ b/open-sse/executors/antigravity/executeAttempt.ts @@ -0,0 +1,687 @@ +// Pure-ish per-attempt request/result helpers for the Antigravity executor (#7408 +// complexity-gate decomposition): building + sending one upstream request, and +// building the final non-streaming/streaming result. No host state of their own — +// callers inject `provider` and `onCreditsUpdate` so this module doesn't need to +// import the executor's credit-balance cache. Extracted from antigravity.ts +// (file-size cap), mirroring the existing antigravity/streamingPassthrough.ts and +// antigravity/sseCollect.ts submodule pattern. +import { mergeAbortSignals, type ExecutorLog } from "../base.ts"; +import { applyFingerprint, isCliCompatEnabled } from "../../config/cliFingerprints.ts"; +import { buildAntigravityUpstreamError } from "../antigravityUpstreamError.ts"; +import { + HTTP_STATUS, + STREAM_READINESS_TIMEOUT_MS, + ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE, +} from "../../config/constants.ts"; +import { injectCreditsField, handleCreditsFailure } from "../../services/antigravityCredits.ts"; +import { cloakAntigravityToolPayload } from "../../config/toolCloaking.ts"; +import { + applyAntigravityClientProfileHeaders, + removeHeaderCaseInsensitive, +} from "../../services/antigravityClientProfile.ts"; +import * as prl from "../../utils/providerRequestLogging.ts"; +import { + createCreditsExtractionTransform as createCreditsExtractionTransformImpl, + buildSsePassthroughResult, + type SsePassthroughResult, +} from "./streamingPassthrough.ts"; +import type { AntigravityCredentials } from "../antigravity.ts"; + +const LONG_RETRY_THRESHOLD_MS = 60_000; +const CREDITS_EXHAUSTED_TTL_MS = 5 * 60 * 60 * 1000; // 5 hours + +/** Invoked with a fresh GOOGLE_ONE_AI credit balance to persist in the caller's cache. */ +export type OnAntigravityCreditsUpdate = (accountId: string, balance: number) => void; + +/** + * Per-account GOOGLE_ONE_AI credits-exhausted tracker. + * Key: accountId (OAuth subject / email). Value: expiry timestamp. + * When credits hit 0 we skip the credit retry for CREDITS_EXHAUSTED_TTL_MS. + * Lives here (not antigravity.ts) so both this module's tryCreditsRetry and + * antigravity.ts's tryResolveRetryFromErrorBody can share it via a single import + * direction (antigravity.ts -> executeAttempt.ts), avoiding a circular import. + */ +const MAX_CREDITS_EXHAUSTED_ENTRIES = 50; +const creditsExhaustedUntil = new Map(); + +const _creditsExhaustedSweep = setInterval(() => { + const now = Date.now(); + for (const [key, until] of creditsExhaustedUntil) { + if (now >= until) creditsExhaustedUntil.delete(key); + } +}, 60_000); +if (typeof _creditsExhaustedSweep === "object" && "unref" in _creditsExhaustedSweep) { + (_creditsExhaustedSweep as { unref?: () => void }).unref?.(); +} + +/** True while `accountId`'s Google One AI credits are marked exhausted. @internal */ +export function isCreditsExhausted(accountId: string): boolean { + const until = creditsExhaustedUntil.get(accountId); + if (!until) return false; + if (Date.now() >= until) { + creditsExhaustedUntil.delete(accountId); + return false; + } + return true; +} + +/** Mark an account's Google One AI credits as exhausted for CREDITS_EXHAUSTED_TTL_MS. */ +export function markCreditsExhausted(accountId: string): void { + if ( + creditsExhaustedUntil.size >= MAX_CREDITS_EXHAUSTED_ENTRIES && + !creditsExhaustedUntil.has(accountId) + ) { + const now = Date.now(); + for (const [key, until] of creditsExhaustedUntil) { + if (now >= until) { + creditsExhaustedUntil.delete(key); + } + } + if (creditsExhaustedUntil.size >= MAX_CREDITS_EXHAUSTED_ENTRIES) { + const oldestKey = creditsExhaustedUntil.keys().next().value; + if (oldestKey !== undefined) creditsExhaustedUntil.delete(oldestKey); + } + } + creditsExhaustedUntil.set(accountId, Date.now() + CREDITS_EXHAUSTED_TTL_MS); +} + +class AntigravityPreResponseTimeoutError extends Error { + code = ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE; + status = HTTP_STATUS.GATEWAY_TIMEOUT; + + constructor(timeoutMs: number, url: string) { + super(`Antigravity upstream did not return response headers within ${timeoutMs}ms: ${url}`); + this.name = "TimeoutError"; + } +} + +function getAbortErrorCode(error: unknown): string | null { + if (!error || typeof error !== "object") return null; + const value = (error as { code?: unknown }).code; + return typeof value === "string" ? value : null; +} + +function isAntigravityPreResponseTimeout(error: unknown): boolean { + return getAbortErrorCode(error) === ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE; +} + +/** + * `fetch()` wrapper that aborts if the upstream never returns response headers + * within `timeoutMs` (default STREAM_READINESS_TIMEOUT_MS) — distinct from the + * overall FETCH_TIMEOUT_MS, which bounds the whole request including body streaming. + * Shared by every fetch attempt in executeOnce() (initial, 403-retry, credits-retry). + */ +export async function fetchAntigravityWithReadinessTimeout( + url: string, + init: RequestInit, + timeoutMs = STREAM_READINESS_TIMEOUT_MS +): Promise { + const boundedTimeoutMs = Math.max(0, Math.floor(timeoutMs)); + if (boundedTimeoutMs <= 0) { + return fetch(url, init); + } + + const timeoutController = new AbortController(); + let timeoutId: ReturnType | null = setTimeout(() => { + timeoutController.abort(new AntigravityPreResponseTimeoutError(boundedTimeoutMs, url)); + }, boundedTimeoutMs); + + const existingSignal = init.signal instanceof AbortSignal ? init.signal : null; + const combinedSignal = existingSignal + ? mergeAbortSignals(existingSignal, timeoutController.signal) + : timeoutController.signal; + + try { + return await fetch(url, { ...init, signal: combinedSignal }); + } catch (error) { + if ( + timeoutController.signal.aborted && + isAntigravityPreResponseTimeout(timeoutController.signal.reason) + ) { + throw timeoutController.signal.reason; + } + throw error; + } finally { + if (timeoutId) { + clearTimeout(timeoutId); + timeoutId = null; + } + } +} + +/** ExecutorLog with every method always callable — see toSafeAntigravityLog(). */ +export type SafeAntigravityLog = Required; + +function noopLogFn(): void {} + +/** + * Normalize a possibly-null/undefined ExecutorLog into an object with all four + * methods always callable, so the request/retry helpers below can call + * `l.debug(...)` directly instead of repeating `log?.debug?.(...)` at every call + * site. This isn't just style: the complexity linter (eslint `complexity` rule) + * weighs each `?.` link in a chain as its own branch — a doubly-chained + * `log?.debug?.(...)` costs +2 — so a logging-heavy helper can rack up a large + * complexity score with zero real decision points. Resolving once here keeps + * the actual branch count legible in the functions that matter. + */ +export function toSafeAntigravityLog(log: ExecutorLog | null | undefined): SafeAntigravityLog { + return { + debug: log?.debug ? log.debug.bind(log) : noopLogFn, + info: log?.info ? log.info.bind(log) : noopLogFn, + warn: log?.warn ? log.warn.bind(log) : noopLogFn, + error: log?.error ? log.error.bind(log) : noopLogFn, + }; +} + +/** Flatten a 429/503 error JSON body (message + `error.details[].reason`) into one string. */ +export function buildAntigravity429ErrorMessage(errorJson: unknown): string { + const obj = errorJson as + | { error?: { message?: unknown; details?: unknown }; message?: unknown } + | null + | undefined; + let errorMessage = String(obj?.error?.message || obj?.message || ""); + const details = obj?.error?.details; + if (Array.isArray(details)) { + for (const detail of details) { + const reason = (detail as { reason?: unknown } | null)?.reason; + if (reason) errorMessage += ` ${reason}`; + } + } + return errorMessage; +} + +function getChunkedOrFixedBody(bodyStr: string, stream: boolean): BodyInit { + if (stream) { + return new ReadableStream( + { + async start(controller) { + controller.enqueue(new TextEncoder().encode(bodyStr)); + controller.close(); + }, + }, + { highWaterMark: 16384 } + ); + } + return bodyStr; +} + +function cloneAntigravityRequestBody(body: unknown): unknown { + if (!body || typeof body !== "object") { + return body; + } + + try { + return structuredClone(body); + } catch { + return JSON.parse(JSON.stringify(body)); + } +} + +function serializeAntigravityRequest( + provider: string, + headers: Record, + body: unknown +): { headers: Record; bodyString: string } { + const serializedBody = cloneAntigravityRequestBody(body); + + if (!isCliCompatEnabled(provider)) { + return { headers, bodyString: JSON.stringify(serializedBody) }; + } + return applyFingerprint(provider, { ...headers }, serializedBody); +} + +function getRequestTargetModel(body: Record): string { + const target = body.model; + return typeof target === "string" && target.length > 0 ? target : "unknown"; +} + +function attachToolNameMap(payload: T, toolNameMap: Map | null): T { + if (!toolNameMap?.size || !payload || typeof payload !== "object") { + return payload; + } + + const copy = Array.isArray(payload) ? ([...payload] as T) : ({ ...(payload as object) } as T); + Object.defineProperty(copy, "_toolNameMap", { + value: toolNameMap, + enumerable: false, + configurable: true, + writable: true, + }); + return copy; +} + +/** Cloak the tool-name payload, then apply credits-first injection, for one attempt. */ +export function finalizeAntigravityRequestBody( + transformed: Record, + useCreditsFirst: boolean, + log: SafeAntigravityLog +): { + transformedBody: Record; + requestToolNameMap: Map | null; +} { + let transformedBody: Record = transformed; + let requestToolNameMap: Map | null = null; + + if (transformedBody && typeof transformedBody === "object") { + const cloaked = cloakAntigravityToolPayload(transformedBody); + transformedBody = cloaked.body; + requestToolNameMap = cloaked.toolNameMap; + } + + // Credits-first: inject GOOGLE_ONE_AI upfront so we never try the normal + // quota path. If credits are exhausted / disabled shouldUseCreditsFirst() + // returns false and we fall back to the legacy retry-on-429 flow. + if (useCreditsFirst) { + transformedBody = injectCreditsField(transformedBody); + log.debug("AG_CREDITS", "Credits-first enabled (ANTIGRAVITY_CREDITS=always)"); + } + + return { transformedBody, requestToolNameMap }; +} + +/** Debug-only dump of outgoing headers (mask Authorization) and envelope shape. */ +function dumpAntigravityRequestDebug( + finalHeaders: Record, + transformedBody: Record, + clientProfile: unknown, + log: SafeAntigravityLog +): void { + const safeHeaders = { ...finalHeaders }; + if (safeHeaders["Authorization"]) safeHeaders["Authorization"] = "Bearer ***"; + log.debug("AG_REQUEST_HEADERS", JSON.stringify(safeHeaders)); + + const envelope = transformedBody as Record; + const requestInner = envelope.request as Record | undefined; + log.debug( + "AG_REQUEST_ENVELOPE", + JSON.stringify({ + fieldOrder: Object.keys(envelope), + project: envelope.project, + requestId: envelope.requestId, + model: envelope.model, + userAgent: envelope.userAgent, + requestType: envelope.requestType, + enabledCreditTypes: envelope.enabledCreditTypes, + clientProfile, + sessionId: requestInner?.sessionId, + generationConfig: requestInner?.generationConfig, + }) + ); +} + +/** + * Send one Antigravity request attempt: serialize + apply the client-profile + * fingerprint, debug-dump the outgoing envelope, fetch with a readiness timeout, + * and transparently retry once without `x-goog-user-project` on a 403 (some + * projects reject that header). Returns the (possibly 403-retried) response + * plus the headers actually used for it. + */ +export async function sendAntigravityRequest( + provider: string, + url: string, + model: string, + headers: Record, + transformedBody: Record, + credentials: AntigravityCredentials, + stream: boolean, + signal: AbortSignal | null | undefined, + log: SafeAntigravityLog, + retryAttempt: number +): Promise<{ response: Response; finalHeaders: Record }> { + const serializedRequest = serializeAntigravityRequest(provider, headers, transformedBody); + let finalHeaders = serializedRequest.headers; + const clientProfile = applyAntigravityClientProfileHeaders( + finalHeaders, + credentials, + transformedBody + ); + + log.debug( + "TELEMETRY", + `[Antigravity] Execute - URL: ${url}, Model: ${model}, Target: ${getRequestTargetModel(transformedBody)}, RetryAttempt: ${retryAttempt}` + ); + + // Dump outgoing headers (mask Authorization) and envelope shape for debugging. + // Gated behind an explicit typeof check (not just calling log.debug() unconditionally) + // so the JSON.stringify work below is skipped entirely when debug logging is off. + if (typeof log.debug === "function") { + dumpAntigravityRequestDebug(finalHeaders, transformedBody, clientProfile, log); + } + + await prl.captureCurrentProviderBody(url, finalHeaders, serializedRequest.bodyString, log); + let response = await fetchAntigravityWithReadinessTimeout(url, { + method: "POST", + headers: finalHeaders, + body: getChunkedOrFixedBody(serializedRequest.bodyString, stream), + ...(stream ? { duplex: "half" } : {}), + signal, + }); + + if (response.status === HTTP_STATUS.FORBIDDEN && finalHeaders["x-goog-user-project"]) { + const retryHeaders = { ...finalHeaders }; + removeHeaderCaseInsensitive(retryHeaders, "x-goog-user-project"); + log.debug("RETRY", "403 with x-goog-user-project, retrying once without it"); + await prl.captureCurrentProviderBody(url, retryHeaders, serializedRequest.bodyString, log); + response = await fetchAntigravityWithReadinessTimeout(url, { + method: "POST", + headers: retryHeaders, + body: getChunkedOrFixedBody(serializedRequest.bodyString, stream), + ...(stream ? { duplex: "half" } : {}), + signal, + }); + finalHeaders = retryHeaders; + } + + if (!response.ok) { + log.warn( + "TELEMETRY", + `[Antigravity] Error Response - URL: ${url}, Status: ${response.status}, Model: ${model}` + ); + } + + return { response, finalHeaders }; +} + +/** + * Retry the SAME url with `enabledCreditTypes: ["GOOGLE_ONE_AI"]` injected, for a + * quota_exhausted 429 that hasn't already tried credits. Returns the result to hand + * back to the caller of execute() on success (or a non-429 status), or null if the + * credits retry also failed/429'd (caller falls through to the normal retry logic). + */ +export async function tryCreditsRetry( + provider: string, + url: string, + headers: Record, + transformedBody: Record, + requestToolNameMap: Map | null, + credentials: AntigravityCredentials, + stream: boolean, + signal: AbortSignal | null | undefined, + log: SafeAntigravityLog, + accountId: string, + onCreditsUpdate: OnAntigravityCreditsUpdate +): Promise { + log.info("AG_CREDITS", "Retrying with Google One AI credits"); + const creditsBody = injectCreditsField(transformedBody); + const serializedCreditsRequest = serializeAntigravityRequest(provider, headers, creditsBody); + const finalCreditsHeaders = serializedCreditsRequest.headers; + try { + await prl.captureCurrentProviderBody( + url, + finalCreditsHeaders, + serializedCreditsRequest.bodyString, + log + ); + const creditsResp = await fetchAntigravityWithReadinessTimeout(url, { + method: "POST", + headers: finalCreditsHeaders, + body: getChunkedOrFixedBody(serializedCreditsRequest.bodyString, stream), + ...(stream ? { duplex: "half" } : {}), + signal, + }); + if (creditsResp.ok || creditsResp.status !== HTTP_STATUS.RATE_LIMITED) { + log.info("AG_CREDITS", `Credits retry succeeded: ${creditsResp.status}`); + if (!stream && creditsResp.body) { + // Raw SSE pass-through + credits extraction (see + // streamingPassthrough.ts); 499s early if the client + // already disconnected instead of piping a cancelled body. + return buildSsePassthroughResult( + creditsResp.body, + creditsResp, + accountId, + onCreditsUpdate, + url, + finalCreditsHeaders, + attachToolNameMap(creditsBody, requestToolNameMap), + signal + ); + } + return { + response: creditsResp, + url, + headers: finalCreditsHeaders, + transformedBody: attachToolNameMap(creditsBody, requestToolNameMap), + }; + } + + // Credit retry also 429'd + handleCreditsFailure(credentials?.accessToken || ""); + log.warn("AG_CREDITS", "Credits retry also 429'd"); + + // Also mark in our legacy exhaustion map to avoid retrying other routes + markCreditsExhausted(accountId); + return null; + } catch (creditsErr) { + handleCreditsFailure(credentials?.accessToken || ""); + log.warn("AG_CREDITS", `Credits retry failed: ${creditsErr}`); + return null; + } +} + +/** + * If we have a 429 with a long retry time (> LONG_RETRY_THRESHOLD_MS), embed + * `retryAfterMs` in the response body so the caller (combo/account-fallback + * layer) can read it back out. Returns null (fall back to the original + * response handling) when the status/retryMs don't qualify, or on error. + */ +export async function tryEmbedLongRetryAfter( + response: Response, + retryMs: number | null, + url: string, + finalHeaders: Record, + transformedBody: Record, + requestToolNameMap: Map | null, + log: ExecutorLog | null | undefined +): Promise { + if ( + response.status !== HTTP_STATUS.RATE_LIMITED || + !retryMs || + retryMs <= LONG_RETRY_THRESHOLD_MS + ) { + return null; + } + try { + const respBody = await response.clone().text(); + let obj; + try { + obj = JSON.parse(respBody); + } catch { + obj = {}; + } + obj.retryAfterMs = retryMs; + const modifiedBody = JSON.stringify(obj); + const modifiedResponse = new Response(modifiedBody, { + status: response.status, + headers: response.headers, + }); + return { + response: modifiedResponse, + url, + headers: finalHeaders, + transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), + }; + } catch (err) { + log?.warn?.("RETRY", `Failed to embed retryAfterMs: ${err}`); + return null; + } +} + +/** Build the sanitized JSON error result shared by the non-streaming and streaming paths. */ +async function buildUpstreamErrorResult( + response: Response, + url: string, + finalHeaders: Record, + transformedBody: Record, + requestToolNameMap: Map | null +): Promise { + const rawBody = await response + .clone() + .text() + .catch(() => ""); + const errorBody = buildAntigravityUpstreamError(response.status, response.statusText, rawBody); + return { + response: new Response(JSON.stringify(errorBody), { + status: response.status, + headers: { "Content-Type": "application/json" }, + }), + url, + headers: finalHeaders, + transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), + }; +} + +/** + * For non-streaming clients, return the raw SSE stream with a + * credits-extraction TransformStream. chatCore's non-streaming path + * (readNonStreamingResponseBody + parseNonStreamingSSEPayload with + * Gemini format support) handles draining and conversion to JSON. + * This replaces the previous collectStreamToResponse() approach which + * had an artificial timeout (now the standard FETCH_BODY_TIMEOUT_MS + * of 10 min applies). + */ +async function buildNonStreamingExecuteOnceResult( + response: Response, + url: string, + finalHeaders: Record, + transformedBody: Record, + requestToolNameMap: Map | null, + accountId: string, + signal: AbortSignal | null | undefined, + onCreditsUpdate: OnAntigravityCreditsUpdate +): Promise { + // #3229: surface a real upstream error instead of masking a 4xx/5xx as an + // empty `chat.completion` envelope. + if (!response.ok) { + return buildUpstreamErrorResult(response, url, finalHeaders, transformedBody, requestToolNameMap); + } + + if (response.body) { + // Raw SSE pass-through + credits extraction (see + // streamingPassthrough.ts); 499s early if the client already + // disconnected instead of piping a cancelled body. + return buildSsePassthroughResult( + response.body, + response, + accountId, + onCreditsUpdate, + url, + finalHeaders, + attachToolNameMap(transformedBody, requestToolNameMap), + signal + ); + } + + // No body -- return as-is + return { + response, + url, + headers: finalHeaders, + transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), + }; +} + +/** + * Streaming path: wrap the response body in a pass-through TransformStream + * that extracts remainingCredits from the final SSE chunk(s) without + * consuming the stream. The client receives the unmodified SSE data. + * + * #2461: a non-ok upstream response (e.g. 403) must never be piped through the + * streaming pass-through below as if it were an SSE body. Google occasionally + * returns non-UTF8/binary error bodies (observed: gzip-magic-byte payloads) for + * 403s on this endpoint; reading/forwarding those raw bytes corrupts the + * client-visible error message. Mirror the non-streaming branch above and build + * a sanitized JSON error via buildAntigravityUpstreamError (hard rule #12) + * instead of streaming unknown bytes straight through. + */ +async function buildStreamingExecuteOnceResult( + response: Response, + url: string, + finalHeaders: Record, + transformedBody: Record, + requestToolNameMap: Map | null, + accountId: string, + signal: AbortSignal | null | undefined, + onCreditsUpdate: OnAntigravityCreditsUpdate +): Promise { + if (!response.ok) { + return buildUpstreamErrorResult(response, url, finalHeaders, transformedBody, requestToolNameMap); + } + + if (response.body) { + // If the downstream client aborts, cancel the upstream fetch body immediately + // to release the socket back to the Undici agent pool and prevent memory leaks. + if (signal) { + const abortHandler = () => { + try { + response.body?.cancel().catch(() => {}); + } catch (_) {} + }; + if (signal.aborted) { + abortHandler(); + } else { + signal.addEventListener("abort", abortHandler, { once: true }); + } + } + + const passThrough = createCreditsExtractionTransformImpl( + accountId, + onCreditsUpdate, + 16 * 1024 // 16KB sliding-window cap to prevent OOM + ); + const tappedBody = response.body.pipeThrough(passThrough); + const tappedResponse = new Response(tappedBody, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }); + return { + response: tappedResponse, + url, + headers: finalHeaders, + transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), + }; + } + + return { + response, + url, + headers: finalHeaders, + transformedBody: attachToolNameMap(transformedBody, requestToolNameMap), + }; +} + +/** Dispatch to the non-streaming or streaming final-result builder. */ +export async function buildFinalAntigravityResult( + stream: boolean, + response: Response, + url: string, + finalHeaders: Record, + transformedBody: Record, + requestToolNameMap: Map | null, + accountId: string, + signal: AbortSignal | null | undefined, + onCreditsUpdate: OnAntigravityCreditsUpdate +): Promise { + if (!stream) { + return buildNonStreamingExecuteOnceResult( + response, + url, + finalHeaders, + transformedBody, + requestToolNameMap, + accountId, + signal, + onCreditsUpdate + ); + } + return buildStreamingExecuteOnceResult( + response, + url, + finalHeaders, + transformedBody, + requestToolNameMap, + accountId, + signal, + onCreditsUpdate + ); +} diff --git a/open-sse/handlers/sseParser/geminiResponse.ts b/open-sse/handlers/sseParser/geminiResponse.ts index d0801019bd2..1657cf20309 100644 --- a/open-sse/handlers/sseParser/geminiResponse.ts +++ b/open-sse/handlers/sseParser/geminiResponse.ts @@ -3,134 +3,149 @@ // state, following the handlers submodule pattern (chatCore/, responseSanitizer/). import { normalizeOpenAICompatibleFinishReasonString } from "../../utils/finishReason.ts"; +type AccumulatedToolCall = { + id: string; + index: number; + type: "function"; + function: { name: string; arguments: string }; +}; + +/** Mutable accumulator threaded through one SSE payload's worth of parsing. */ +type GeminiSSEAccumulator = { + textContent: string; + finishReason: string; + usage: Record | null; + sawContent: boolean; + toolCalls: AccumulatedToolCall[]; +}; + +function stripZeroWidth(value: unknown): unknown { + if (typeof value === "string") return value.replace(/[\u200B-\u200D\uFEFF]/g, ""); + return value; +} + /** - * Convert Gemini/Antigravity SSE chunks into a single non-streaming OpenAI - * chat.completion JSON response. Gemini SSE carries payloads like: - * - * data: {"markdown":"...chunk..."} - * data: {"response":{"candidates":[{"content":{"parts":[{"text":"..."}]},"finishReason":"STOP"}],"usageMetadata":{...}}} - * data: {"remainingCredits":[...]} - * - * Reuses the same parsing logic as processAntigravitySSEPayload() in sseCollect.ts - * so that format conversion is functionally equivalent to the previous - * collectStreamToResponse() approach. Intentional differences: - * - remainingCredits is NOT embedded into the result (handled separately - * by the credits-extraction TransformStream in antigravity.ts). - * - The synthetic `id` uses `chatcmpl-${Date.now()}` (no UUID suffix) - * because this path runs once per response, not per chunk. + * Detect the `[Tool call: name]\nArguments: {...}` textual convention some + * Gemini/Antigravity models emit instead of a native functionCall part. */ -export function parseSSEToGeminiResponse( - rawSSE: string, - fallbackModel: string -): Record | null { - const lines = String(rawSSE || "").split("\n"); - let textContent = ""; - let finishReason = "stop"; - let usage: Record | null = null; - let sawContent = false; - - type AccumulatedToolCall = { - id: string; - index: number; - type: "function"; - function: { name: string; arguments: string }; - }; - const toolCalls: AccumulatedToolCall[] = []; +function tryParseTextualToolCall(text: string): { name: string; args: unknown } | null { + const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, ""); + const match = normalized.match( + /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/ + ); + if (!match) return null; + const name = match[1]?.trim(); + const rawArgs = match[2]?.trim(); + if (!name || !rawArgs) return null; + try { + return { name, args: stripZeroWidth(JSON.parse(rawArgs)) }; + } catch { + return null; + } +} - const stripZeroWidth = (value: unknown): unknown => { - if (typeof value === "string") return value.replace(/[\u200B-\u200D\uFEFF]/g, ""); - return value; - }; +/** Extract the markdown shortcut some Gemini variants send (top-level or nested). */ +function extractGeminiMarkdownShortcut(parsed: Record): string | null { + if (typeof parsed.markdown === "string") return parsed.markdown; + const response = parsed.response as Record | undefined; + return typeof response?.markdown === "string" ? response.markdown : null; +} - const tryParseTextualToolCall = (text: string): { name: string; args: unknown } | null => { - const normalized = text.replace(/[\u200B-\u200D\uFEFF]/g, ""); - const match = normalized.match( - /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/ - ); - if (!match) return null; - const name = match[1]?.trim(); - const rawArgs = match[2]?.trim(); - if (!name || !rawArgs) return null; - try { - return { name, args: stripZeroWidth(JSON.parse(rawArgs)) }; - } catch { - return null; - } +/** Append one candidate content part (text or textual tool call) onto the accumulator. */ +function applyCandidatePart(part: Record, acc: GeminiSSEAccumulator): void { + if (typeof part.text !== "string" || part.thought || part.thoughtSignature) return; + + const textualToolCall = tryParseTextualToolCall(part.text); + if (textualToolCall) { + acc.toolCalls.push({ + id: `${textualToolCall.name}-${Date.now()}-${acc.toolCalls.length}`, + index: acc.toolCalls.length, + type: "function", + function: { + name: textualToolCall.name, + arguments: JSON.stringify(textualToolCall.args || {}), + }, + }); + } else { + acc.textContent += part.text; + } + acc.sawContent = true; +} + +/** Walk the first candidate's content parts, if present, mutating the accumulator. */ +function applyCandidateContentParts( + candidate: Record | undefined, + acc: GeminiSSEAccumulator +): void { + const content = candidate?.content as Record | undefined; + const parts = content?.parts; + if (!Array.isArray(parts)) return; + for (const part of parts) { + applyCandidatePart(part as Record, acc); + } +} + +/** Normalize and apply the candidate's finishReason, if present. */ +function applyFinishReason( + candidate: Record | undefined, + acc: GeminiSSEAccumulator +): void { + if (!candidate?.finishReason) return; + acc.finishReason = normalizeOpenAICompatibleFinishReasonString( + String(candidate.finishReason).toLowerCase() + ); +} + +/** Extract usageMetadata into the OpenAI-shaped usage object, if present. */ +function applyUsageMetadata(parsed: Record, acc: GeminiSSEAccumulator): void { + const response = parsed.response as Record | undefined; + const um = response?.usageMetadata as Record | undefined; + if (!um) return; + acc.usage = { + prompt_tokens: um.promptTokenCount || 0, + completion_tokens: um.candidatesTokenCount || 0, + total_tokens: um.totalTokenCount || 0, }; +} - for (const line of lines) { - const trimmed = line.trim(); - if (!trimmed.startsWith("data:")) continue; - const payload = trimmed.slice(5).trim(); - if (!payload || payload === "[DONE]") continue; +/** Parse one `data:` line's JSON payload and fold it into the accumulator (best-effort). */ +function applyGeminiSSEDataLine(payload: string, acc: GeminiSSEAccumulator): void { + try { + const parsed = JSON.parse(payload) as Record; - try { - const parsed = JSON.parse(payload); - - // Markdown shortcut (some Gemini variants) - const markdown = - typeof parsed?.markdown === "string" - ? parsed.markdown - : typeof parsed?.response?.markdown === "string" - ? parsed.response.markdown - : null; - if (markdown) { - textContent += markdown; - sawContent = true; - } - - // Candidate content parts - const candidate = parsed?.response?.candidates?.[0]; - if (candidate?.content?.parts) { - for (const part of candidate.content.parts) { - if (typeof part.text === "string" && !part.thought && !part.thoughtSignature) { - const textualToolCall = tryParseTextualToolCall(part.text); - if (textualToolCall) { - toolCalls.push({ - id: `${textualToolCall.name}-${Date.now()}-${toolCalls.length}`, - index: toolCalls.length, - type: "function", - function: { - name: textualToolCall.name, - arguments: JSON.stringify(textualToolCall.args || {}), - }, - }); - } else { - textContent += part.text; - } - sawContent = true; - } - } - } - - if (candidate?.finishReason) { - finishReason = normalizeOpenAICompatibleFinishReasonString( - String(candidate.finishReason).toLowerCase() - ); - } - - if (parsed?.response?.usageMetadata) { - const um = parsed.response.usageMetadata; - usage = { - prompt_tokens: um.promptTokenCount || 0, - completion_tokens: um.candidatesTokenCount || 0, - total_tokens: um.totalTokenCount || 0, - }; - } - } catch { - // Ignore malformed lines + const markdown = extractGeminiMarkdownShortcut(parsed); + if (markdown) { + acc.textContent += markdown; + acc.sawContent = true; } - } - if (!sawContent && toolCalls.length === 0) return null; + const response = parsed.response as Record | undefined; + const candidates = response?.candidates; + const candidate = Array.isArray(candidates) + ? (candidates[0] as Record | undefined) + : undefined; + + applyCandidateContentParts(candidate, acc); + applyFinishReason(candidate, acc); + applyUsageMetadata(parsed, acc); + } catch { + // Ignore malformed lines + } +} +/** Assemble the final non-streaming chat.completion payload from the accumulator. */ +function buildChatCompletionFromAccumulator( + acc: GeminiSSEAccumulator, + fallbackModel: string +): Record { const message: Record = { role: "assistant", - content: textContent || null, + content: acc.textContent || null, }; - if (toolCalls.length > 0) { - message.tool_calls = toolCalls; + let finishReason = acc.finishReason; + if (acc.toolCalls.length > 0) { + message.tool_calls = acc.toolCalls; finishReason = "tool_calls"; } @@ -148,9 +163,52 @@ export function parseSSEToGeminiResponse( ], }; - if (usage) { - result.usage = usage; + if (acc.usage) { + result.usage = acc.usage; } return result; } + +/** + * Convert Gemini/Antigravity SSE chunks into a single non-streaming OpenAI + * chat.completion JSON response. Gemini SSE carries payloads like: + * + * data: {"markdown":"...chunk..."} + * data: {"response":{"candidates":[{"content":{"parts":[{"text":"..."}]},"finishReason":"STOP"}],"usageMetadata":{...}}} + * data: {"remainingCredits":[...]} + * + * Reuses the same parsing logic as processAntigravitySSEPayload() in sseCollect.ts + * so that format conversion is functionally equivalent to the previous + * collectStreamToResponse() approach. Intentional differences: + * - remainingCredits is NOT embedded into the result (handled separately + * by the credits-extraction TransformStream in antigravity.ts). + * - The synthetic `id` uses `chatcmpl-${Date.now()}` (no UUID suffix) + * because this path runs once per response, not per chunk. + */ +export function parseSSEToGeminiResponse( + rawSSE: string, + fallbackModel: string +): Record | null { + const lines = String(rawSSE || "").split("\n"); + const acc: GeminiSSEAccumulator = { + textContent: "", + finishReason: "stop", + usage: null, + sawContent: false, + toolCalls: [], + }; + + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + + applyGeminiSSEDataLine(payload, acc); + } + + if (!acc.sawContent && acc.toolCalls.length === 0) return null; + + return buildChatCompletionFromAccumulator(acc, fallbackModel); +} From b3054d19cc47638ad801de6dead20e9670bfc2a1 Mon Sep 17 00:00:00 2001 From: Minxi Hou Date: Sat, 18 Jul 2026 16:05:32 -0300 Subject: [PATCH 13/14] Merge branch 'release/v3.8.49' into fix/antigravity-streaming-passthrough MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves conflict in open-sse/executors/antigravity.ts between this branch's streaming-passthrough decomposition and #7290's fallback-chain decomposition (already merged into release/v3.8.49) — both sides added imports from the same new antigravity/ submodule files, kept both. Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> --- .env.example | 4 + .github/workflows/ci.yml | 36 +- .github/workflows/codeql.yml | 4 +- .github/workflows/dast-smoke.yml | 2 +- .github/workflows/electron-release.yml | 2 +- .github/workflows/mutation-redundancy.yml | 2 +- .github/workflows/nightly-compat.yml | 4 +- .github/workflows/nightly-llm-security.yml | 4 +- .github/workflows/nightly-mutation.yml | 4 +- .github/workflows/nightly-property.yml | 2 +- .github/workflows/nightly-release-green.yml | 4 +- .github/workflows/nightly-resilience.yml | 8 +- .github/workflows/nightly-schemathesis.yml | 2 +- .github/workflows/npm-publish.yml | 4 +- .github/workflows/opencode-plugin-ci.yml | 4 +- .github/workflows/opencode-provider-ci.yml | 4 +- .github/workflows/quality.yml | 14 +- .github/workflows/wiki-sync.yml | 2 +- README.md | 6 +- .../features/7318-router-eval-harness.md | 1 + .../provider-quota-connection-visibility.md | 3 + .../fixes/7032-auggie-model-ids-v032.md | 1 + changelog.d/fixes/7490-align-engines-node.md | 1 + .../fixes/7547-prefer-public-endpoint-url.md | 1 + .../fixes/pending-cli-service-detection.md | 1 + .../fixes/pending-react-flow-dark-theme.md | 1 + .../7334-incident-response-runbook.md | 1 + .../7336-perf-latency-budgets-doc.md | 1 + config/quality/complexity-baseline.json | 3 +- config/quality/dependency-allowlist.json | 1 + config/quality/file-size-baseline.json | 14 +- config/quality/test-discovery-baseline.json | 1 - docs/INCIDENT_RESPONSE.md | 194 +++ docs/PERF_BUDGETS.md | 227 +++ docs/compression/COMPRESSION_GUIDE.md | 2 + docs/compression/RTK_COMPRESSION.md | 52 +- docs/frameworks/MCP-SERVER.md | 30 +- docs/openapi.yaml | 30 + docs/reference/ENVIRONMENT.md | 1 + docs/reference/PROVIDER_PLUGIN_MANIFEST.md | 8 + .../00_SESSION_OVERVIEW.md | 36 + .../01_RESEARCH.md | 28 + .../02_SPECIFICATIONS.md | 43 + .../03_DAG_WBS.md | 25 + .../04_IMPLEMENTATION_STRATEGY.md | 37 + .../05_KNOWN_ISSUES.md | 21 + .../06_TESTING_STRATEGY.md | 25 + electron/package-lock.json | 12 +- electron/package.json | 2 +- next.config.mjs | 4 +- open-sse/config/agyModels.ts | 4 +- open-sse/config/antigravityModelAliases.ts | 6 +- open-sse/config/freeModelCatalog.data.ts | 1 + open-sse/config/imageRegistry.ts | 61 +- .../config/nvidiaHostedModels.snapshot.json | 18 + open-sse/config/providers/index.ts | 2 + .../config/providers/registry/agnes/index.ts | 26 + .../config/providers/registry/auggie/index.ts | 54 +- .../config/providers/registry/claude/index.ts | 2 +- .../providers/registry/gemini/imageModels.ts | 32 + .../providers/registry/opencode/go/index.ts | 14 +- .../registry/stability-ai/imageModels.ts | 76 + open-sse/executors/antigravity.ts | 69 +- .../executors/antigravity/proFallbackChain.ts | 104 ++ open-sse/executors/antigravity/sseCollect.ts | 16 +- open-sse/executors/auggie.ts | 147 +- open-sse/executors/base/reasoningEffort.ts | 124 +- open-sse/executors/default.ts | 27 +- open-sse/executors/duckduckgo-web.ts | 65 + open-sse/executors/opencode.ts | 38 +- open-sse/executors/theoldllm.ts | 216 ++- open-sse/handlers/chatCore.ts | 147 +- .../chatCore/cavemanOutputAnalytics.ts | 8 +- .../chatCore/compressionAnalyticsWrite.ts | 61 +- .../handlers/chatCore/outputTokenBudget.ts | 117 ++ .../handlers/chatCore/streamingPipeline.ts | 21 +- open-sse/handlers/imageGeneration.ts | 12 + .../imageGeneration/providers/googleImagen.ts | 147 ++ open-sse/mcp-server/README.md | 16 +- .../__tests__/toolSearch.catalog.test.ts | 9 +- open-sse/mcp-server/schemas/ccrTools.ts | 202 +++ open-sse/mcp-server/schemas/index.ts | 20 + open-sse/mcp-server/schemas/tools.ts | 3 + open-sse/mcp-server/server.ts | 1 + open-sse/mcp-server/toolSearch/catalog.ts | 5 +- open-sse/mcp-server/tools/compressionTools.ts | 259 ++- open-sse/package.json | 3 +- open-sse/services/accountFallback.ts | 15 +- .../accountFallback/lockoutEviction.ts | 53 + open-sse/services/claudeCodeCompatible.ts | 1 + open-sse/services/combo.ts | 81 +- open-sse/services/combo/autoConfig.ts | 4 +- open-sse/services/combo/comboPredicates.ts | 44 +- open-sse/services/combo/comboStructure.ts | 71 +- .../services/combo/knownContextOverflow.ts | 97 ++ .../services/combo/resolveAutoStrategy.ts | 14 +- open-sse/services/combo/sessionStickiness.ts | 34 + open-sse/services/combo/targetExhaustion.ts | 17 +- open-sse/services/combo/targetSorters.ts | 48 +- open-sse/services/combo/validateQuality.ts | 26 +- open-sse/services/compression/bodyAdapter.ts | 67 +- .../services/compression/engines/ccr/index.ts | 347 +++- .../compression/engines/ionizer/sample.ts | 19 +- .../compression/engines/rtk/filterLoader.ts | 167 +- .../compression/engines/rtk/filterSchema.ts | 6 + .../compression/engines/rtk/lineFilter.ts | 48 + .../engines/rtk/tomlCompatibility.ts | 334 ++++ .../engines/session-dedup/fuzzy.ts | 11 +- open-sse/services/compression/liveZone.ts | 397 +++++ open-sse/services/compression/types.ts | 13 + open-sse/services/responsesInputSanitizer.ts | 26 +- .../translator/request/claude-to-gemini.ts | 4 +- .../request/openai-responses/toResponses.ts | 37 + open-sse/utils/diagnostics.ts | 22 + open-sse/utils/sseHeartbeat.ts | 21 + open-sse/utils/stream.ts | 18 + package-lock.json | 1513 ++++++++--------- package.json | 10 +- scripts/check/check-nvidia-catalog-drift.ts | 108 ++ scripts/check/check-router-eval-regression.ts | 287 ++++ scripts/router-eval/compare.ts | 135 ++ scripts/router-eval/index.ts | 483 ++++++ scripts/router-eval/patch-compare.ts | 315 ++++ scripts/router-eval/search.ts | 439 +++++ scripts/router-eval/trends.ts | 198 +++ skills/omni-context-rtk/SKILL.md | 11 + .../dashboard/HomeProviderTopologySection.tsx | 18 +- .../analytics/CompressionAnalyticsTab.tsx | 10 +- src/app/(dashboard)/dashboard/combos/page.tsx | 5 +- .../context/rtk/RtkContextPageClient.tsx | 8 +- .../context/rtk/RtkTomlImportCard.tsx | 255 +++ .../context/settings/CompressionPanel.tsx | 57 +- .../dashboard/endpoint/EndpointPageClient.tsx | 56 +- .../__tests__/ApiEndpointsTab.test.tsx | 16 +- .../__tests__/EndpointPageClient.test.tsx | 52 + .../(dashboard)/dashboard/onboarding/page.tsx | 11 +- .../[id]/ProviderDetailPageClient.tsx | 7 +- .../providers/[id]/__tests__/phase1f.test.tsx | 6 +- .../[id]/components/ConnectionRow.tsx | 11 + .../[id]/components/ConnectionsListPanel.tsx | 22 + .../ProviderQuotaVisibilityToggle.tsx | 43 + .../components/__tests__/phase1d.test.tsx | 37 +- .../[id]/hooks/useProviderConnections.ts | 7 +- .../[id]/hooks/useProviderQuotaVisibility.ts | 45 + .../settings/components/ProxyTab.tsx | 16 +- .../settings/components/SidebarTab.tsx | 65 +- .../settings/components/proxy/FreePoolTab.tsx | 84 +- .../dashboard/usage/components/EvalsTab.tsx | 6 +- .../ProviderLimits/CodexResetCreditsModal.tsx | 277 +++ .../components/ProviderLimits/QuotaCard.tsx | 19 +- .../ProviderLimits/QuotaCardGrid.tsx | 9 +- .../usage/components/ProviderLimits/index.tsx | 17 +- .../parts/QuotaCardExpanded.tsx | 52 +- .../useCodexResetCreditRedemption.ts | 207 ++- .../(dashboard)/home/ProviderQuotaWidget.tsx | 3 + src/app/api/cli-tools/all-statuses/route.ts | 17 +- src/app/api/context/rtk/import/route.ts | 81 + src/app/api/issue-agent/runs/route.ts | 141 ++ src/app/api/network/info/route.ts | 3 +- .../api/oauth/[provider]/[action]/route.ts | 7 +- .../models/discovery/providerModelsConfig.ts | 24 + src/app/api/providers/[id]/route.ts | 7 +- .../agent-bridge/agents/[id]/dns/route.ts | 4 +- src/app/api/usage/codex-reset-credit/route.ts | 65 +- src/app/api/v1/images/generations/route.ts | 9 +- src/app/api/v1/models/catalog.ts | 56 +- src/app/api/v1/models/catalogHelpers.ts | 57 +- .../api/v1/provider-plugin-manifest/route.ts | 42 +- src/app/globals.css | 22 +- src/i18n/messages/de.json | 69 +- src/i18n/messages/en.json | 43 +- src/i18n/messages/pt-BR.json | 43 +- src/instrumentation-node.ts | 58 + src/lib/cliTools/batchStatusCache.ts | 24 +- src/lib/config/runtimeSettings.ts | 9 +- src/lib/credentialHealth/scheduler.ts | 9 +- src/lib/db/backup.ts | 8 +- src/lib/db/caseMapping.ts | 15 +- src/lib/db/cleanup.ts | 104 +- src/lib/db/compression.ts | 4 + src/lib/db/core.ts | 21 +- src/lib/db/databaseSettings.ts | 3 + src/lib/db/migrationRunner.ts | 6 +- ...5_provider_connection_quota_visibility.sql | 5 + src/lib/db/providers.ts | 168 +- src/lib/db/proxies.ts | 83 +- src/lib/db/proxies/guards.ts | 81 + src/lib/db/schemaColumns.ts | 6 + src/lib/initCloudSync.ts | 9 +- src/lib/issueAgent/audit.ts | 43 + src/lib/issueAgent/execution.ts | 114 ++ src/lib/issueAgent/githubExport.ts | 57 + src/lib/issueAgent/recordedTriage.ts | 124 ++ src/lib/localHealthCheck.ts | 9 +- src/lib/oauth/constants/oauth.ts | 6 +- src/lib/oauth/providers/codebuddy-cn.ts | 5 +- src/lib/providers/validation/webProvidersB.ts | 19 +- src/lib/quota/connectionRecovery.ts | 9 +- src/lib/resilience/modelLockoutSettings.ts | 7 +- src/lib/routerEval/index.ts | 428 +++++ src/lib/tokenHealthCheck.ts | 43 +- src/lib/usage/codexResetCredits.ts | 182 +- src/lib/usage/comboScoringInspector.ts | 111 +- src/mitm/dns/provision.ts | 128 +- src/server/authz/routeGuard.ts | 1 + src/server/ws/liveServer.ts | 6 +- src/shared/components/Sidebar.tsx | 47 +- src/shared/components/flow/FlowCanvas.tsx | 1 + src/shared/components/flow/edgeStyles.ts | 4 +- src/shared/constants/mcpScopes.ts | 6 + .../constants/providers/apikey/regional.ts | 12 + src/shared/constants/sidebarVisibility.ts | 5 +- .../constants/sidebarVisibility/sections.ts | 1 - .../constants/sidebarVisibility/types.ts | 10 +- .../__tests__/useDisplayBaseUrl.test.tsx | 108 +- src/shared/hooks/cli/useToolBatchStatuses.ts | 11 +- src/shared/hooks/index.ts | 7 +- src/shared/hooks/useDisplayBaseUrl.ts | 122 +- src/shared/services/cliRuntime.ts | 8 +- src/shared/types/utilization.ts | 12 +- src/shared/utils/providerQuotaVisibility.ts | 13 + src/shared/utils/sidebarExpansionState.ts | 29 + src/shared/utils/testProcess.ts | 44 + .../validation/compressionConfigSchemas.ts | 1 + src/shared/validation/schemas/provider.ts | 1 + src/sse/handlers/chat.ts | 15 +- src/sse/handlers/comboFailureLogging.ts | 20 + src/types/databaseSettings.ts | 6 + stryker.conf.json | 1 + tests/fixtures/router-eval/baseline.ndjson | 2 + tests/fixtures/router-eval/candidate.ndjson | 2 + tests/integration/all-statuses-route.test.ts | 44 +- tests/theoldllm-stress.test.ts | 72 +- .../account-fallback-lockout-eviction.test.ts | 129 ++ .../unit/agent-bridge-dns-params-7271.test.ts | 37 + .../agentrouter-models-discovery-7016.test.ts | 74 + tests/unit/agnes-provider.test.ts | 70 + .../unit/agy-pro-fallback-chain-3786.test.ts | 203 +++ tests/unit/antigravity-model-aliases.test.ts | 10 +- ...antigravity-server-side-tools-6914.test.ts | 43 + .../compression/rtk-toml-import-route.test.ts | 125 ++ .../v1/provider-plugin-manifest-route.test.ts | 34 +- .../apikey-connection-health-check.test.ts | 74 + tests/unit/auggie-executor.test.ts | 170 +- .../auto-combo-context-advertising.test.ts | 8 +- .../autoCombo/provider-family-combos.test.ts | 7 +- .../base-executor-sanitize-effort.test.ts | 35 + tests/unit/batch-status-cache.test.ts | 8 + ...atcore-compression-analytics-write.test.ts | 20 +- tests/unit/chatcore-sanitization.test.ts | 2 +- .../unit/chatcore-streaming-pipeline.test.ts | 39 +- ...claude-context-1m-supported-models.test.ts | 153 ++ ...-to-gemini-budget-tokens-zero-6813.test.ts | 42 + tests/unit/cli-runtime-detection.test.ts | 56 +- tests/unit/clinepass-thinking-budget.test.ts | 16 +- tests/unit/codex-reset-credits.test.ts | 147 ++ tests/unit/combo-auto-config-split.test.ts | 17 + tests/unit/combo-breaker-429.test.ts | 11 + .../unit/combo-context-window-filter.test.ts | 204 ++- tests/unit/combo-failure-log-message.test.ts | 29 + tests/unit/combo-least-used-account.test.ts | 86 + .../combo-resolve-auto-strategy-split.test.ts | 136 ++ tests/unit/combo-scoring-inspector.test.ts | 114 +- ...bo-stickiness-responses-input-7270.test.ts | 235 +++ .../combo/combo-target-exhaustion.test.ts | 19 + tests/unit/compression/body-adapter.test.ts | 86 + .../unit/compression/ccr-cross-tenant.test.ts | 18 +- .../compression/ccr-mcp-integration.test.ts | 220 +++ .../compression-preview-api.test.ts | 4 + tests/unit/compression/db.test.ts | 11 + tests/unit/compression/live-zone.test.ts | 254 +++ .../unit/compression/rtk-line-filter.test.ts | 16 + .../rtk-toml-compatibility.test.ts | 313 ++++ tests/unit/db-core-temp-store-pragma.test.ts | 21 + tests/unit/db-providers-crud.test.ts | 95 ++ tests/unit/db-schema-columns-split.test.ts | 22 + ...uit-breaker-null-content-6999-7000.test.ts | 248 +++ tests/unit/diagnostics.test.ts | 23 +- tests/unit/executor-agy.test.ts | 85 + tests/unit/executor-antigravity.test.ts | 2 +- tests/unit/free-models.test.ts | 4 + tests/unit/gemini-imagen-predict.test.ts | 86 + .../unit/image-text-to-image-modality.test.ts | 145 ++ ...instrumentation-warm-catalog-cache.test.ts | 162 ++ tests/unit/issue-agent-audit.test.ts | 31 + tests/unit/issue-agent-execution.test.ts | 70 + tests/unit/issue-agent-github-export.test.ts | 48 + .../unit/issue-agent-route-execution.test.ts | 135 ++ tests/unit/issue-agent-runner.test.ts | 63 + tests/unit/issue-agent-runs-route.test.ts | 285 ++++ .../m365-web-token-extraction-7078.test.ts | 48 + tests/unit/mcp-tool-count-dedup-6854.test.ts | 11 +- .../mitm-dns-graceful-degrade-6127.test.ts | 127 +- .../models-catalog-combo-metadata.test.ts | 119 ++ tests/unit/models-catalog-route.test.ts | 2 +- tests/unit/node-runtime-support.test.ts | 9 + tests/unit/nvidia-catalog-drift.test.ts | 28 + .../unit/oauth-device-code-endpoints.test.ts | 32 + ...uth-device-code-error-transparency.test.ts | 32 + .../openai-responses-request-split.test.ts | 67 + .../opencode-go-effort-aliases-6922.test.ts | 164 ++ tests/unit/output-token-budget.test.ts | 66 + ...ovider-connections-quota-threshold.test.ts | 23 + tests/unit/provider-limits-ui.test.ts | 14 + tests/unit/provider-quota-visibility.test.ts | 18 + .../unit/proxy-bulk-import-dedup-7594.test.ts | 105 ++ .../responses-input-sanitizer-name.test.ts | 58 +- tests/unit/route-guard-private-lan.test.ts | 5 + tests/unit/router-eval-check.test.ts | 484 ++++++ tests/unit/router-eval-cli.test.ts | 270 +++ tests/unit/router-eval-compare.test.ts | 83 + tests/unit/router-eval-e2e-chain.test.ts | 135 ++ tests/unit/router-eval-patch-compare.test.ts | 312 ++++ tests/unit/router-eval-search.test.ts | 258 +++ tests/unit/router-eval-trends.test.ts | 128 ++ tests/unit/router-eval.test.ts | 92 + tests/unit/sseHeartbeat.test.ts | 75 + ...stream-request-body-size-mark-7045.test.ts | 122 ++ .../unit/telemetry-auto-cleanup-6848.test.ts | 214 +++ tests/unit/test-process-detection.test.ts | 52 + tests/unit/theoldllm-provider-proxy.test.ts | 60 + tests/unit/token-health-check-sweep.test.ts | 146 ++ tests/unit/ui/combos-page-smoke.test.tsx | 12 + tests/unit/ui/edgeStyles.test.ts | 6 +- tests/unit/ui/evals-tab-smoke.test.tsx | 12 + tests/unit/ui/flowCanvas.test.tsx | 13 +- tests/unit/{ => ui}/free-pool-tab.test.tsx | 37 +- ...me-provider-topology-section-4606.test.tsx | 14 +- .../ui/onboarding-public-endpoint.test.tsx | 107 ++ tests/unit/ui/rtkTomlImportCard.test.tsx | 179 ++ .../unit/ui/sidebar-proxy-expansion.test.tsx | 46 + tests/unit/ui/useToolBatchStatuses.test.tsx | 11 +- 332 files changed, 19674 insertions(+), 2135 deletions(-) create mode 100644 changelog.d/features/7318-router-eval-harness.md create mode 100644 changelog.d/features/provider-quota-connection-visibility.md create mode 100644 changelog.d/fixes/7032-auggie-model-ids-v032.md create mode 100644 changelog.d/fixes/7490-align-engines-node.md create mode 100644 changelog.d/fixes/7547-prefer-public-endpoint-url.md create mode 100644 changelog.d/fixes/pending-cli-service-detection.md create mode 100644 changelog.d/fixes/pending-react-flow-dark-theme.md create mode 100644 changelog.d/maintenance/7334-incident-response-runbook.md create mode 100644 changelog.d/maintenance/7336-perf-latency-budgets-doc.md create mode 100644 docs/INCIDENT_RESPONSE.md create mode 100644 docs/PERF_BUDGETS.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/00_SESSION_OVERVIEW.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/01_RESEARCH.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/02_SPECIFICATIONS.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/03_DAG_WBS.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/04_IMPLEMENTATION_STRATEGY.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/05_KNOWN_ISSUES.md create mode 100644 docs/sessions/20260714-issue-agent-executable-triage/06_TESTING_STRATEGY.md create mode 100644 open-sse/config/nvidiaHostedModels.snapshot.json create mode 100644 open-sse/config/providers/registry/agnes/index.ts create mode 100644 open-sse/config/providers/registry/gemini/imageModels.ts create mode 100644 open-sse/config/providers/registry/stability-ai/imageModels.ts create mode 100644 open-sse/executors/antigravity/proFallbackChain.ts create mode 100644 open-sse/handlers/chatCore/outputTokenBudget.ts create mode 100644 open-sse/handlers/imageGeneration/providers/googleImagen.ts create mode 100644 open-sse/mcp-server/schemas/ccrTools.ts create mode 100644 open-sse/services/accountFallback/lockoutEviction.ts create mode 100644 open-sse/services/combo/knownContextOverflow.ts create mode 100644 open-sse/services/compression/engines/rtk/tomlCompatibility.ts create mode 100644 open-sse/services/compression/liveZone.ts create mode 100644 scripts/check/check-nvidia-catalog-drift.ts create mode 100644 scripts/check/check-router-eval-regression.ts create mode 100644 scripts/router-eval/compare.ts create mode 100644 scripts/router-eval/index.ts create mode 100644 scripts/router-eval/patch-compare.ts create mode 100644 scripts/router-eval/search.ts create mode 100644 scripts/router-eval/trends.ts create mode 100644 src/app/(dashboard)/dashboard/context/rtk/RtkTomlImportCard.tsx create mode 100644 src/app/(dashboard)/dashboard/providers/[id]/components/ProviderQuotaVisibilityToggle.tsx create mode 100644 src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderQuotaVisibility.ts create mode 100644 src/app/(dashboard)/dashboard/usage/components/ProviderLimits/CodexResetCreditsModal.tsx create mode 100644 src/app/api/context/rtk/import/route.ts create mode 100644 src/app/api/issue-agent/runs/route.ts create mode 100644 src/lib/db/migrations/125_provider_connection_quota_visibility.sql create mode 100644 src/lib/db/proxies/guards.ts create mode 100644 src/lib/issueAgent/audit.ts create mode 100644 src/lib/issueAgent/execution.ts create mode 100644 src/lib/issueAgent/githubExport.ts create mode 100644 src/lib/issueAgent/recordedTriage.ts create mode 100644 src/lib/routerEval/index.ts create mode 100644 src/shared/utils/providerQuotaVisibility.ts create mode 100644 src/shared/utils/sidebarExpansionState.ts create mode 100644 src/shared/utils/testProcess.ts create mode 100644 src/sse/handlers/comboFailureLogging.ts create mode 100644 tests/fixtures/router-eval/baseline.ndjson create mode 100644 tests/fixtures/router-eval/candidate.ndjson create mode 100644 tests/unit/account-fallback-lockout-eviction.test.ts create mode 100644 tests/unit/agent-bridge-dns-params-7271.test.ts create mode 100644 tests/unit/agentrouter-models-discovery-7016.test.ts create mode 100644 tests/unit/agnes-provider.test.ts create mode 100644 tests/unit/antigravity-server-side-tools-6914.test.ts create mode 100644 tests/unit/api/compression/rtk-toml-import-route.test.ts create mode 100644 tests/unit/claude-context-1m-supported-models.test.ts create mode 100644 tests/unit/claude-to-gemini-budget-tokens-zero-6813.test.ts create mode 100644 tests/unit/combo-failure-log-message.test.ts create mode 100644 tests/unit/combo-least-used-account.test.ts create mode 100644 tests/unit/combo-stickiness-responses-input-7270.test.ts create mode 100644 tests/unit/compression/ccr-mcp-integration.test.ts create mode 100644 tests/unit/compression/live-zone.test.ts create mode 100644 tests/unit/compression/rtk-toml-compatibility.test.ts create mode 100644 tests/unit/db-core-temp-store-pragma.test.ts create mode 100644 tests/unit/ddg-circuit-breaker-null-content-6999-7000.test.ts create mode 100644 tests/unit/gemini-imagen-predict.test.ts create mode 100644 tests/unit/image-text-to-image-modality.test.ts create mode 100644 tests/unit/instrumentation-warm-catalog-cache.test.ts create mode 100644 tests/unit/issue-agent-audit.test.ts create mode 100644 tests/unit/issue-agent-execution.test.ts create mode 100644 tests/unit/issue-agent-github-export.test.ts create mode 100644 tests/unit/issue-agent-route-execution.test.ts create mode 100644 tests/unit/issue-agent-runner.test.ts create mode 100644 tests/unit/issue-agent-runs-route.test.ts create mode 100644 tests/unit/m365-web-token-extraction-7078.test.ts create mode 100644 tests/unit/models-catalog-combo-metadata.test.ts create mode 100644 tests/unit/nvidia-catalog-drift.test.ts create mode 100644 tests/unit/oauth-device-code-endpoints.test.ts create mode 100644 tests/unit/oauth-device-code-error-transparency.test.ts create mode 100644 tests/unit/opencode-go-effort-aliases-6922.test.ts create mode 100644 tests/unit/output-token-budget.test.ts create mode 100644 tests/unit/provider-quota-visibility.test.ts create mode 100644 tests/unit/proxy-bulk-import-dedup-7594.test.ts create mode 100644 tests/unit/router-eval-check.test.ts create mode 100644 tests/unit/router-eval-cli.test.ts create mode 100644 tests/unit/router-eval-compare.test.ts create mode 100644 tests/unit/router-eval-e2e-chain.test.ts create mode 100644 tests/unit/router-eval-patch-compare.test.ts create mode 100644 tests/unit/router-eval-search.test.ts create mode 100644 tests/unit/router-eval-trends.test.ts create mode 100644 tests/unit/router-eval.test.ts create mode 100644 tests/unit/sseHeartbeat.test.ts create mode 100644 tests/unit/stream-request-body-size-mark-7045.test.ts create mode 100644 tests/unit/telemetry-auto-cleanup-6848.test.ts create mode 100644 tests/unit/test-process-detection.test.ts create mode 100644 tests/unit/theoldllm-provider-proxy.test.ts create mode 100644 tests/unit/token-health-check-sweep.test.ts create mode 100644 tests/unit/ui/combos-page-smoke.test.tsx create mode 100644 tests/unit/ui/evals-tab-smoke.test.tsx rename tests/unit/{ => ui}/free-pool-tab.test.tsx (90%) create mode 100644 tests/unit/ui/onboarding-public-endpoint.test.tsx create mode 100644 tests/unit/ui/rtkTomlImportCard.test.tsx create mode 100644 tests/unit/ui/sidebar-proxy-expansion.test.tsx diff --git a/.env.example b/.env.example index b4b73bfd3a3..fdb38b67f7f 100644 --- a/.env.example +++ b/.env.example @@ -649,6 +649,10 @@ NEXT_PUBLIC_ENABLE_SOCKS5_PROXY=true # Legacy alias for OMNIROUTE_API_KEY. # ROUTER_API_KEY= +# Enable the offline/local Issue Agent recorded-triage endpoint. +# Used by: src/app/api/issue-agent/runs/route.ts. Default: disabled. +# OMNIROUTE_ISSUE_AGENT_ENABLED=false + # CLI remote-mode context/profile for `omniroute` commands (overrides the active # context in the local contexts store). Equivalent to the `--context ` flag. # Used by: bin/cli/program.mjs, bin/cli/api.mjs (remote mode). diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d97161e470c..71c4d640b46 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -39,7 +39,7 @@ jobs: with: persist-credentials: false fetch-depth: 0 - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} - id: classify @@ -82,7 +82,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -171,7 +171,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -280,7 +280,7 @@ jobs: with: persist-credentials: false fetch-depth: 0 - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -388,7 +388,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -424,7 +424,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -452,7 +452,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -515,7 +515,7 @@ jobs: with: persist-credentials: false fetch-depth: 0 - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} - name: Fetch base branch @@ -554,7 +554,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -602,7 +602,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -649,7 +649,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -710,7 +710,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -757,7 +757,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -802,7 +802,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -871,7 +871,7 @@ jobs: # (if-no-files-found: warn) — Sonar consumes the same file. - name: Upload coverage to Codecov (informational) if: always() - uses: codecov/codecov-action@04b047e8bb82a0c002c8312c1c880fbc6a999d45 # v5 + uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5 with: files: coverage/lcov.info token: ${{ secrets.CODECOV_TOKEN }} @@ -1045,7 +1045,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -1115,7 +1115,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -1138,7 +1138,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index e47f93b3f69..34dd99666aa 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -22,10 +22,10 @@ jobs: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: persist-credentials: false - - uses: github/codeql-action/init@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0 + - uses: github/codeql-action/init@7188fc363630916deb702c7fdcf4e481b751f97a # v4.37.1 with: languages: javascript-typescript queries: security-extended - - uses: github/codeql-action/analyze@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0 + - uses: github/codeql-action/analyze@7188fc363630916deb702c7fdcf4e481b751f97a # v4.37.1 with: category: "/language:javascript-typescript" diff --git a/.github/workflows/dast-smoke.yml b/.github/workflows/dast-smoke.yml index e1b5c757e55..08a622fdbba 100644 --- a/.github/workflows/dast-smoke.yml +++ b/.github/workflows/dast-smoke.yml @@ -21,7 +21,7 @@ jobs: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: persist-credentials: false - - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 with: node-version: "24" cache: npm diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml index b86cd50f69f..2a7fee0e9ae 100644 --- a/.github/workflows/electron-release.yml +++ b/.github/workflows/electron-release.yml @@ -88,7 +88,7 @@ jobs: with: persist-credentials: false - name: Setup Node - uses: actions/setup-node@v6 + uses: actions/setup-node@v7 with: node-version: 24 cache: npm diff --git a/.github/workflows/mutation-redundancy.yml b/.github/workflows/mutation-redundancy.yml index e584bea4870..0960846c5fe 100644 --- a/.github/workflows/mutation-redundancy.yml +++ b/.github/workflows/mutation-redundancy.yml @@ -44,7 +44,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm diff --git a/.github/workflows/nightly-compat.yml b/.github/workflows/nightly-compat.yml index 2abbb149d94..eabc2073b32 100644 --- a/.github/workflows/nightly-compat.yml +++ b/.github/workflows/nightly-compat.yml @@ -66,7 +66,7 @@ jobs: with: ref: ${{ needs.resolve-branch.outputs.target }} persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "26" cache: npm @@ -93,7 +93,7 @@ jobs: with: ref: ${{ needs.resolve-branch.outputs.target }} persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ matrix.node }} cache: npm diff --git a/.github/workflows/nightly-llm-security.yml b/.github/workflows/nightly-llm-security.yml index f039690235b..6a9abac2948 100644 --- a/.github/workflows/nightly-llm-security.yml +++ b/.github/workflows/nightly-llm-security.yml @@ -15,7 +15,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: { node-version: "24", cache: npm } - run: npm ci - name: Build CLI bundle @@ -65,7 +65,7 @@ jobs: with: persist-credentials: false if: steps.gate.outputs.run == 'true' - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 if: steps.gate.outputs.run == 'true' with: { node-version: "24", cache: npm } - run: npm ci diff --git a/.github/workflows/nightly-mutation.yml b/.github/workflows/nightly-mutation.yml index 5dfbe345a7f..5ee377deb96 100644 --- a/.github/workflows/nightly-mutation.yml +++ b/.github/workflows/nightly-mutation.yml @@ -107,7 +107,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm @@ -151,7 +151,7 @@ jobs: - uses: actions/checkout@v6 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" - name: Download all mutation reports diff --git a/.github/workflows/nightly-property.yml b/.github/workflows/nightly-property.yml index 776426573d4..2ef7a543aff 100644 --- a/.github/workflows/nightly-property.yml +++ b/.github/workflows/nightly-property.yml @@ -13,7 +13,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm diff --git a/.github/workflows/nightly-release-green.yml b/.github/workflows/nightly-release-green.yml index 63d1b4a054f..b9b5ddc41ad 100644 --- a/.github/workflows/nightly-release-green.yml +++ b/.github/workflows/nightly-release-green.yml @@ -116,7 +116,7 @@ jobs: git checkout "$TARGET" git log -1 --oneline - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm @@ -228,7 +228,7 @@ jobs: fetch-depth: 0 persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm diff --git a/.github/workflows/nightly-resilience.yml b/.github/workflows/nightly-resilience.yml index af5a20e22da..f6adffeabf3 100644 --- a/.github/workflows/nightly-resilience.yml +++ b/.github/workflows/nightly-resilience.yml @@ -15,7 +15,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm @@ -29,7 +29,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm @@ -43,7 +43,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm @@ -94,7 +94,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "24" cache: npm diff --git a/.github/workflows/nightly-schemathesis.yml b/.github/workflows/nightly-schemathesis.yml index e430677638a..64adef8f50b 100644 --- a/.github/workflows/nightly-schemathesis.yml +++ b/.github/workflows/nightly-schemathesis.yml @@ -16,7 +16,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: { node-version: "24", cache: npm } - run: npm ci - name: Build CLI bundle diff --git a/.github/workflows/npm-publish.yml b/.github/workflows/npm-publish.yml index 1edd3c09e08..692e6bd26c9 100644 --- a/.github/workflows/npm-publish.yml +++ b/.github/workflows/npm-publish.yml @@ -71,7 +71,7 @@ jobs: fetch-depth: 0 - name: Setup Node.js - uses: actions/setup-node@v6 + uses: actions/setup-node@v7 with: node-version: ${{ env.NPM_PUBLISH_NODE_VERSION }} registry-url: https://registry.npmjs.org @@ -265,7 +265,7 @@ jobs: # Full history needed for auto-bump: git diff against previous release tag - name: Setup Node.js - uses: actions/setup-node@v6 + uses: actions/setup-node@v7 with: node-version: ${{ env.NPM_PUBLISH_NODE_VERSION }} registry-url: https://registry.npmjs.org diff --git a/.github/workflows/opencode-plugin-ci.yml b/.github/workflows/opencode-plugin-ci.yml index 0925d4c7f8f..f1c99fe426c 100644 --- a/.github/workflows/opencode-plugin-ci.yml +++ b/.github/workflows/opencode-plugin-ci.yml @@ -35,7 +35,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ matrix.node }} cache: npm @@ -52,7 +52,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "22" cache: npm diff --git a/.github/workflows/opencode-provider-ci.yml b/.github/workflows/opencode-provider-ci.yml index 52a206dd374..5fe7d1fe78d 100644 --- a/.github/workflows/opencode-provider-ci.yml +++ b/.github/workflows/opencode-provider-ci.yml @@ -35,7 +35,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ matrix.node }} cache: npm @@ -51,7 +51,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: "20" cache: npm diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index e54c4ac4c25..87adbc6b0f4 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -36,7 +36,7 @@ jobs: with: persist-credentials: false fetch-depth: 0 - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} - id: classify @@ -68,7 +68,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -99,7 +99,7 @@ jobs: with: fetch-depth: 0 persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -225,7 +225,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -265,7 +265,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -299,7 +299,7 @@ jobs: - uses: actions/checkout@v7 with: persist-credentials: false - - uses: actions/setup-node@v6 + - uses: actions/setup-node@v7 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm @@ -343,7 +343,7 @@ jobs: with: fetch-depth: 0 persist-credentials: false - - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6 + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v6 with: node-version: ${{ env.CI_NODE_VERSION }} cache: npm diff --git a/.github/workflows/wiki-sync.yml b/.github/workflows/wiki-sync.yml index da22cccd8d1..9ef2cdee2d5 100644 --- a/.github/workflows/wiki-sync.yml +++ b/.github/workflows/wiki-sync.yml @@ -40,7 +40,7 @@ jobs: uses: actions/checkout@v7 - name: Setup Node - uses: actions/setup-node@v6 + uses: actions/setup-node@v7 with: node-version: "24" diff --git a/README.md b/README.md index c6c656b921f..9559dd0a88a 100644 --- a/README.md +++ b/README.md @@ -892,7 +892,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo | OAuth token expired | Auto-refreshed; if stuck, delete + re-auth in Providers | | `unsupported_country_region_territory` | Configure proxy in Settings → Proxy | | Docker SQLite locks | Use `--stop-timeout 40` for clean WAL checkpoint | -| Node runtime errors | Use Node `>=22.0.0 <23` or `>=24.0.0 <27` | +| Node runtime errors | Use Node `>=22.22.2 <23` or `>=24.0.0 <27` | 🐛 **Reporting a bug?** Run `npm run system-info` and attach `system-info.txt`. 📖 [`docs/guides/TROUBLESHOOTING.md`](docs/guides/TROUBLESHOOTING.md) @@ -936,7 +936,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo -- **Runtime**: Node.js 22.x or 24.x LTS (24 LTS recommended) — `>=22.0.0 <23 || >=24.0.0 <27` +- **Runtime**: Node.js 22.x or 24.x LTS (24 LTS recommended) — `>=22.22.2 <23 || >=24.0.0 <27` - **Language**: TypeScript 6.0 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) - **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 - **Database**: better-sqlite3 (SQLite) + LowDB (JSON legacy) — domain state, proxy logs, MCP audit, routing decisions, memory, skills @@ -1251,7 +1251,7 @@ MIT License - see [LICENSE](LICENSE) for details. **[⬆ Back to top](#-omniroute)** · Built with ❤️ for the open-source AI community. -OmniRoute v3.8.43 · Node ≥22.0.0 · MIT License · omniroute.online +OmniRoute v3.8.43 · Node ≥22.22.2 · MIT License · omniroute.online diff --git a/changelog.d/features/7318-router-eval-harness.md b/changelog.d/features/7318-router-eval-harness.md new file mode 100644 index 00000000000..f88326e9e01 --- /dev/null +++ b/changelog.d/features/7318-router-eval-harness.md @@ -0,0 +1 @@ +- **feat(eval):** added a router-eval harness (`npm run eval:router`, `eval:router:compare`, `eval:router:patch-compare`, `eval:router:search`, `eval:router:trends`, `check:router-eval`) that replays routing decisions — from NDJSON corpora or the `usage_history`/`call_logs` SQLite tables — into an AIQ (success/latency/cost) score, compares baseline vs. candidate router configs with a retained-run regression gate, and ranks Pareto-optimal candidates; a sibling tool to the existing `eval:compression` harness (#7318 — thanks @KooshaPari). diff --git a/changelog.d/features/provider-quota-connection-visibility.md b/changelog.d/features/provider-quota-connection-visibility.md new file mode 100644 index 00000000000..46e18fd848b --- /dev/null +++ b/changelog.d/features/provider-quota-connection-visibility.md @@ -0,0 +1,3 @@ +- Provider connections can now be shown or hidden individually on the Provider Quota page. The + visibility setting is available on each account in the provider detail view and does not affect + routing or account activation. diff --git a/changelog.d/fixes/7032-auggie-model-ids-v032.md b/changelog.d/fixes/7032-auggie-model-ids-v032.md new file mode 100644 index 00000000000..bec9929b7b2 --- /dev/null +++ b/changelog.d/fixes/7032-auggie-model-ids-v032.md @@ -0,0 +1 @@ +- **fix(providers):** Auggie (Augment CLI) model registry updated to the real v0.32.0 CLI model IDs, with a live `auggie model list` auto-discovery fallback for future renames; the old pre-v0.32.0 IDs (`claude-sonnet-4.6`, `claude-opus-4.6`, `claude-haiku-4.5`, `gemini-3.1-pro`, `gemini-3.0-flash`, the `gpt-5.4`/`gpt-5.5` high/medium variants) are a **breaking rename**, but a backward-compat alias map in `resolveAuggieModel()` transparently remaps them to their v0.32.0 equivalents so existing saved combos keep working without any manual update (#7032 — thanks @oyi77). diff --git a/changelog.d/fixes/7490-align-engines-node.md b/changelog.d/fixes/7490-align-engines-node.md new file mode 100644 index 00000000000..2b6839f7746 --- /dev/null +++ b/changelog.d/fixes/7490-align-engines-node.md @@ -0,0 +1 @@ +- fix(build): align `engines.node` supported range (>=22.22.2 <23 || >=24.0.0 <27) across package.json, lockfile, README engine references and the node-runtime support test, so install-time engine checks match the actually-tested runtimes (#7446) diff --git a/changelog.d/fixes/7547-prefer-public-endpoint-url.md b/changelog.d/fixes/7547-prefer-public-endpoint-url.md new file mode 100644 index 00000000000..387f0d09a1a --- /dev/null +++ b/changelog.d/fixes/7547-prefer-public-endpoint-url.md @@ -0,0 +1 @@ +- **fix(dashboard):** Public and managed tunnel endpoints now take precedence over loopback URLs in dashboard setup and copyable API configuration ([#7547](https://github.com/diegosouzapw/OmniRoute/pull/7547)) — thanks @nguyenha935 diff --git a/changelog.d/fixes/pending-cli-service-detection.md b/changelog.d/fixes/pending-cli-service-detection.md new file mode 100644 index 00000000000..f9752264d4d --- /dev/null +++ b/changelog.d/fixes/pending-cli-service-detection.md @@ -0,0 +1 @@ +- **fix(cli):** CLI detection now refreshes stale cached results, reports discovered versions, and checks the Continue `cn` binary instead of assuming it is installed. diff --git a/changelog.d/fixes/pending-react-flow-dark-theme.md b/changelog.d/fixes/pending-react-flow-dark-theme.md new file mode 100644 index 00000000000..1e177ee7359 --- /dev/null +++ b/changelog.d/fixes/pending-react-flow-dark-theme.md @@ -0,0 +1 @@ +- **fix(ui):** Theme React Flow controls correctly in dark mode, improve idle connector contrast, and localize the provider topology legend. diff --git a/changelog.d/maintenance/7334-incident-response-runbook.md b/changelog.d/maintenance/7334-incident-response-runbook.md new file mode 100644 index 00000000000..32a2286cffa --- /dev/null +++ b/changelog.d/maintenance/7334-incident-response-runbook.md @@ -0,0 +1 @@ +- **docs:** Add `docs/INCIDENT_RESPONSE.md` — a non-security incident-response runbook (severity ladder, first-15-minutes checklist, and per-failure-mode mitigation steps for provider outages, latency regressions, and auth/data-layer incidents) ([#7334](https://github.com/diegosouzapw/OmniRoute/pull/7334)) — thanks @KooshaPari diff --git a/changelog.d/maintenance/7336-perf-latency-budgets-doc.md b/changelog.d/maintenance/7336-perf-latency-budgets-doc.md new file mode 100644 index 00000000000..a74a2a670a0 --- /dev/null +++ b/changelog.d/maintenance/7336-perf-latency-budgets-doc.md @@ -0,0 +1 @@ +- **docs:** Add `docs/PERF_BUDGETS.md` — per-endpoint p50/p95/p99 latency, throughput, resource, and cold-start budget reference targets ([#7336](https://github.com/diegosouzapw/OmniRoute/pull/7336)) — thanks @KooshaPari diff --git a/config/quality/complexity-baseline.json b/config/quality/complexity-baseline.json index 6917dbb5e41..262870277a8 100644 --- a/config/quality/complexity-baseline.json +++ b/config/quality/complexity-baseline.json @@ -1,6 +1,7 @@ { "_comment": "Catraca de complexidade (check-complexity.mjs, ESLint core rules complexity>=15 e max-lines-per-function>80 sobre src+open-sse+electron+bin via eslint.complexity.config.mjs). Conta total de violacoes; so pode cair. --update ratcheta.", - "count": 2058, + "count": 2059, + "_rebaseline_2026_07_18_pr7360_quota_visibility_resync": "2058->2059 (+1 vs recorded ceiling; measured 2056 fresh on release tip cab9e5f0c alone, so this ceiling still carries 2 units of un-banked slack from prior shrinkage — real regression from this merge is 2056->2059, +3). PR #7360 (JxnLexn) release-resync: merging origin/release/v3.8.49 to resolve the 3-file conflict (ConnectionRow.tsx/ConnectionsListPanel.tsx/useProviderConnections.ts) unions two already-compliant features in the same already-oversized god-component: release's confirm-delete-account wiring (#7361) and this PR's per-connection quota-visibility wiring. Diffed release-tip-only vs merged violation lists (scripts dumped via getComplexityEslintReport): most entries are the SAME pre-existing violations shifted a few lines (ConnectionRow/getStatusPresentation/inferErrorType — no count change) or marginally bigger (ConnectionRow function complexity 85->86, ConnectionsListPanel function 498->510 lines) from the two ConnectionRow call sites each gaining both PRs' multi-line JSX props. The 2 genuinely NEW crossings are the 'no tag' and 'tagged groups' .map() render callbacks in ConnectionsListPanel.tsx (83 and 85 lines, was <=80 on both parents individually) tipping over 80 lines specifically because both PRs' props land on the same call sites. No new logic was written during the resync itself (only import-statement unions); the growth is inherent to combining the two already-reviewed feature branches. Structural shrink tracked in #3501. Tighten via --update next cycle (true floor is 2056, not 2058).", "_rebaseline_2026_07_17_v3849_ownerprs_providers": "2056->2058 (+2). v3.8.49 owner-PR merge campaign own-growth: the new provider handlers/dispatch branches merged this cycle (freetheai/felo/notion/segmind/deepinfra/novita/msdesigner image+video handlers, each adding a format-dispatch guard) pushed cyclomatic violations 2056->2058. Fast-gates PR->release do not run the complexity ratchet, so this surfaced only on re-sync. Spread across the new leaf handlers (not a single extractable function); measured on the release tip. Structural shrink tracked in #3501.", "_rebaseline_2026_07_10_v3847_merge_burst": "2053->2054 (+1). Drift herdado do merge burst do dia em release/v3.8.47 (campanha /implement-prs: ~36 PRs mergeados — órfãos, features do dono, ports). O check:complexity NÃO roda no fast-path PR->release, então o ramo acumulou o +1 sem rebaselinar (mesma família de todos os rebaselines abaixo). Trust-but-verify: medido 2054 no tip da release pós-burst; a única função flagada nova é pré-existente (getResolvedModelCapabilities em modelCapabilities.ts, já >teto antes de #6714). Nenhum PR órfão/feature introduz violação NOVA — os fixes deste ciclo são complexity-net-zero. Rebaseline aprovado pelo dono (2026-07-10) para destravar o FQG dos ~7 órfãos verdes-exceto-complexity. Tighten via --update next cycle.", "_rebaseline_2026_07_10_gcf_v3_2": "2054->2056 (+2). PR feat/headroom-gcf-v3.2-nested-flattening: own growth from re-vendoring the GCF (Headroom) codec to spec v3.2 (nested flattening). The 2 new over-threshold functions are the v3.2 `>`-path flatten/unflatten walk in the vendored generic-profile encode/decode paths (open-sse/services/compression/engines/headroom/gcf/{generic,decode_generic}.ts). This is imported third-party code kept byte-faithful to upstream gcf-typescript, not extractable without diverging from the vendored source; local measures 2055 on the merged tree; frozen at 2056 = the base's CI-observed 2054 + this PR's 2 new functions, matching the documented local-vs-CI off-by-one convention (see _rebaseline_2026_07_02_v3844_ci_observed) so the GitHub runner stays green. Round-trip guarded by tests/unit/compression/headroom-smartcrusher.test.ts (deep-nested case). Structural shrink belongs upstream in gcf, not here.", diff --git a/config/quality/dependency-allowlist.json b/config/quality/dependency-allowlist.json index 5ecfb4329a5..6c7e2e3f452 100644 --- a/config/quality/dependency-allowlist.json +++ b/config/quality/dependency-allowlist.json @@ -114,6 +114,7 @@ "safe-regex", "selfsigned", "size-limit", + "smol-toml", "socks", "sql.js", "sqlite-vec", diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 46a56405b99..0c7d6db2a5f 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_07_18_pr7360_quota_visibility_resync": "PR #7360 (JxnLexn, per-connection Provider Quota visibility) release-resync: merging origin/release/v3.8.49 into the PR's own branch to resolve its 3-file conflict (ConnectionRow.tsx/ConnectionsListPanel.tsx/useProviderConnections.ts import unions) surfaced ProviderDetailPageClient.tsx 785->791 (measured via check-file-size.mjs's split('\\n').length; wc -l reports 790). Both parents were individually within the 786 cap: release tip alone measures 785 (its own confirm-delete-account #7361 wiring), and the PR's own pre-merge tip also measured 785 (its own quota-visibility wiring) — the two independent, already-cohesive features simply sum past the frozen ceiling once unioned in this same already-oversized god-component (tracked for decomposition under #3501). No new logic was written during the resync; the only edits were import-statement unions. Frozen raised 786->791.", "_rebaseline_2026_07_18_pr7653_chat_tracker_import": "PR #7653 merge-interaction growth: release moved chat.ts to its 1796 cap while this PR adds the single side-effect import 'quotaTrackersBatch.ts' (line 130) — chat.ts IS the canonical quota-fetcher registration point (codex/bailian/deepseek/openrouter/opencode/generic all import+register there), so the +1 is irreducible call-site wiring. 1796->1797. Covered by tests/unit/{agentrouter,v0,freemodel}-quota-fetcher.test.ts.", "_rebaseline_2026_07_17_pr7653_agentrouter_console_fields": "PR #7653 own growth (missing acceptance criterion: the AgentRouter quota tracker (#6850) read providerSpecificData.consoleApiKey/newApiUserId but neither field had dashboard UI for provider agentrouter — consoleApiKey was gated to bailian-coding-plan only and newApiUserId had zero UI). AddApiKeyModal.tsx 961->967 (+6) and EditConnectionModal.tsx 1278->1286 (+8) = import + a single render call plus the newApiUserId formData init field. The actual Input rendering (both consoleApiKey reuse + the new newApiUserId field) was EXTRACTED into a new leaf src/app/(dashboard)/dashboard/providers/[id]/components/modals/AgentrouterConsoleFields.tsx (48 LOC, 2462 (+1, irreducible at the existing model-aware preflight chokepoint — the `provider === \"codex\"` check that forwards requestedModel into the connection arg is extended to also cover `openrouter`, one added boolean + a doc comment, offset to a single net line by dropping the now-redundant inline condition). Enforcement itself lives in open-sse/services/openrouterQuotaFetcher.ts (not frozen) and the dispatch-time record/correct hooks live in open-sse/executors/base.ts (not frozen). Covered by tests/unit/openrouter-free-window-wiring-6842.test.ts.", @@ -142,6 +143,7 @@ "_rebaseline_2026_06_20_1449_1444_test_route": "Re-baseline providers test route.ts 842->887: combined growth of sibling fixes #1449 (bound OAuth connection-test probe with a timeout) + #1444 (label a deactivated account distinctly from a revoked token), both at the same connection-test chokepoint. Cohesive route handler; not extractable without hiding the test flow.", "_rebaseline_2026_06_20_1409_1294_models": "Re-baseline src/lib/db/models.ts 1184->1221: combined growth of sibling fixes #1409 (cascade-delete orphaned model aliases when a provider is removed) + #1294 (persist max_input_tokens/max_output_tokens on custom models), both adding CRUD at the existing models domain module. Cohesive db module; not extractable.", "_rebaseline_2026_06_20_4389_thinking_toolchoice": "Re-baseline base.ts 1387->1399 (#4389): tool_choice-forced thinking guard at the existing Claude wire-image injection chokepoint (effThinking gate avoids the Anthropic 400 when tool_choice forces a tool). Cohesive guard; structural shrink tracked in #3501.", + "_rebaseline_2026_07_18_6979_codex_test": "PR #6979 own growth: executor-codex.test.ts 1340->1347 (+7 = generalized ensureThinkingBudget assertion added to the existing codex thinking-budget cases). antigravity-test bump 942->977 REVERTED here: #7408's test split dropped that file to 888, so this PR's +35 fits under the original 942 frozen cap.", "cap": 800, "frozen": { "_rebaseline_2026_07_02_5816_qoder": "PR #5816 (@AgentKiller45, qoder PAT via qodercli): qoderCli.ts 666->989, new-above-cap frozen (owner-approved baseline freeze). The growth is the legitimate PAT job-token exchange + quota parsing CLI transport (the pure-JS Cosy path 500'd on every PAT request); extracting the spawn/parse helpers now would just add indirection to a contributor PR mid-merge. Test frozen also raised for this PR's coverage growth: providers-page-utils.test.ts 1052->1092. Additionally clears an inherited base-red from the already-merged #5933 (codex json_schema->text.format): translator-openai-responses-req.test.ts 1097->1172 (+75 regression tests, no offending branch left). All remain frozen (cannot grow further); release captain's rebaseline-at-release supersedes.", @@ -198,7 +200,7 @@ "open-sse/translator/request/openai-to-kiro.ts": 912, "open-sse/translator/response/openai-responses.ts": 1092, "open-sse/utils/cursorAgentProtobuf.ts": 1521, - "open-sse/utils/stream.ts": 2796, + "open-sse/utils/stream.ts": 2814, "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1385, "src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx": 1028, "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 3120, @@ -206,13 +208,14 @@ "src/app/(dashboard)/dashboard/cache/page.tsx": 845, "src/app/(dashboard)/dashboard/cli-code/components/CodexToolCard.tsx": 900, "src/app/(dashboard)/dashboard/cloud-agents/page.tsx": 922, - "src/app/(dashboard)/dashboard/combos/page.tsx": 4655, + "_rebaseline_2026_07_15_7070_combos_memo": "PR #7070 (perf/p1-memo) own growth: src/app/(dashboard)/dashboard/combos/page.tsx 4655->4656 (+1 = React.memo wrapping of ComboCard). Covered by tests/unit/ui/combos-page-smoke.test.tsx.", + "src/app/(dashboard)/dashboard/combos/page.tsx": 4656, "src/app/(dashboard)/dashboard/costs/CostOverviewTab.tsx": 1495, "src/app/(dashboard)/dashboard/costs/quota-share/components/PoolWizard.tsx": 1007, "src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.tsx": 2612, "src/app/(dashboard)/dashboard/health/page.tsx": 1091, "src/app/(dashboard)/dashboard/playground/components/tabs/ApiTab.tsx": 847, - "src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx": 786, + "src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx": 791, "src/app/(dashboard)/dashboard/providers/[id]/components/ConnectionRow.tsx": 942, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 967, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": 1286, @@ -314,7 +317,7 @@ "tests/unit/db-settings-crud.test.ts": 941, "tests/unit/deepseek-web.test.ts": 1092, "tests/unit/executor-antigravity.test.ts": 942, - "tests/unit/executor-codex.test.ts": 1340, + "tests/unit/executor-codex.test.ts": 1347, "tests/unit/executor-default-base.test.ts": 1523, "tests/unit/grok-web.test.ts": 2437, "tests/unit/image-generation-handler.test.ts": 2019, @@ -408,5 +411,6 @@ "_rebaseline_2026_07_07_6534_chirag": "PR #6534 (@chirag127) own growth: open-sse/services/compression/strategySelector.ts ->1025. Owner-approved rebaseline. Frozen.", "_rebaseline_2026_07_08_6556_omniglyph_mode": "PR #6556 (omniglyph engine) own growth: open-sse/services/compression/strategySelector.ts 1025->1043 (+18 at the existing mode-dispatch chokepoints). Two single-mode branches (sync no-op + async resolve via the engine registry, mirroring the rtk single-mode pattern, B-MODE-ENGINE-DECOUPLE) plus the optional providerTransport field threaded through the three options types (gates transport-sensitive engines). The engine itself lives in engines/omniglyphAdapter.ts (876. Owner-approved rebaseline. Frozen.", - "_rebaseline_2026_07_07_6525_chirag_image_guard": "PR #6525 (@chirag127, #6457) own growth: chat.ts ->1778 (reject image-only models on /v1/chat/completions; stacks on #6515). Owner-approved. Frozen." + "_rebaseline_2026_07_07_6525_chirag_image_guard": "PR #6525 (@chirag127, #6457) own growth: chat.ts ->1778 (reject image-only models on /v1/chat/completions; stacks on #6515). Owner-approved. Frozen.", + "_rebaseline_2026_07_15_7045_perf_instrumentation": "PR #7045 (@oyi77) own growth: open-sse/utils/stream.ts 2796->2814 (+18) from performance.mark/measure instrumentation around the SSE dispatch chokepoint (b48ba21c4), a TextEncoder hoisting fix to avoid a per-chunk allocation on the hot path (c35e8a9b4), and clearing the fixed-name \"omni-request-body-size\" mark immediately after creation (babysit fix, addressing a review-flagged unbounded-growth leak in Node's global performance timeline). Cohesive wiring at the existing stream-dispatch chokepoint; not extractable. Covered by tests/unit/chatcore-streaming-pipeline.test.ts + tests/unit/stream-request-body-size-mark-7045.test.ts." } diff --git a/config/quality/test-discovery-baseline.json b/config/quality/test-discovery-baseline.json index baca8262965..e6cb0d2a49f 100644 --- a/config/quality/test-discovery-baseline.json +++ b/config/quality/test-discovery-baseline.json @@ -42,7 +42,6 @@ "tests/unit/dashboard/batch/list-regression.test.tsx", "tests/unit/dashboard/batch/sanitization.test.tsx", "tests/unit/free-budget-card.test.tsx", - "tests/unit/free-pool-tab.test.tsx", "tests/unit/guardrails/visionBridgeRouter.test.tsx", "tests/unit/omni-skills-page.test.tsx", "tests/unit/shared-clipboard.test.tsx", diff --git a/docs/INCIDENT_RESPONSE.md b/docs/INCIDENT_RESPONSE.md new file mode 100644 index 00000000000..95cada8e84b --- /dev/null +++ b/docs/INCIDENT_RESPONSE.md @@ -0,0 +1,194 @@ +# Incident Response Runbook — OmniRoute (2026-06-18) + +**Status**: Authoritative. The 71-pillar audit (L61) references this doc +for the `Obs > 2.00` gate. +**Owner**: observability-circle (lead: security-circle lead). +**SLOs**: see `docs/PERF_BUDGETS.md` § 1 (top-level SLOs) and +`ops/slos.yaml` (machine-readable form, generated by the Bifrost team). +**Disclosure policy**: see `SECURITY.md` (vulnerability disclosure only, +separate flow). + +This runbook is the operational playbook for **non-security** incidents: +outages, latency regressions, error-budget burn, and provider-side +failures. Vulnerability disclosure stays on `SECURITY.md`; do not route +those through this runbook. + +--- + +## 1. Severity ladder + +| Sev | Definition | Examples | Page on | Resolve by | +|---|---|---|---|---| +| **SEV-1** | User-visible outage; > 50 % of requests failing or > 2x SLO breach for 5 min. | Cluster down; auth layer broken; 5xx flood. | On-call P0 (immediate) | 4 h | +| **SEV-2** | Significant degradation; 1.5–2x SLO breach for 15 min, or single-tenant impact. | Single provider down; p95 > 1.5x budget; rate-limit runaway. | On-call P1 (15 min) | 24 h | +| **SEV-3** | Latent bug or near-miss; no current user impact but error budget at risk. | Memory leak trending up; circuit breaker tripping on one provider. | Slack `#omniroute-ops` (next standup) | 7 d | +| **SEV-4** | Cosmetic / informational. | Log line noise; non-binding UI glitch. | Next weekly review | Next refactor cycle | + +**Burn-rate escalation** (per `docs/PERF_BUDGETS.md` § 1): 6x for 5 min +is SEV-1; 2x for 1 h is SEV-2; sustained < 1x for 7 d demotes to SEV-3. + +--- + +## 2. Detection sources + +| Source | Signal | Routing | +|---|---|---| +| Prometheus (`/metrics`) | Counter deltas (5xx, latency) | Alertmanager → PagerDuty | +| Grafana SLO dashboards | SLO burn-rate panels | Slack `#omniroute-ops` | +| Uptime probe (`/api/health/ping`) | 3 consecutive failures from 3 regions | Alertmanager → PagerDuty | +| Dependabot | New CVE in dependency | GitHub issue + Slack `#security` | +| User report (support@) | Manual triage | Slack `#omniroute-triage` | +| Error budget burn alert | `slo_burn_rate > threshold` | Alertmanager | + +Prometheus and Alertmanager are configured in the deploy repo (see +`docs/operations/DEPLOY.md` once published; currently inline in +`docker-compose.prod.yml`). + +--- + +## 3. First-15-minutes checklist + +When paged, the on-call engineer runs this checklist verbatim. **Do +not** skip steps; each is timed. + +1. **0:00** — Acknowledge the page in PagerDuty. Stops the escalation + timer and notifies the secondary. +2. **0:02** — Open the [SLO dashboard][dash] and the [incident + channel][chan] (`#inc-YYYY-MM-DD-slug`). Post a single-line ack + with the alert name and the time. +3. **0:05** — Classify severity per § 1. If SEV-1 or SEV-2, declare + the incident in the channel and tag `@incident-commander`. +4. **0:08** — Capture the alert payload, the most recent deploy SHA, + and the top 5 slow / erroring endpoints. Post to the channel. +5. **0:12** — Decide: **mitigate first, root-cause later**. Choose + one of: + - **Roll back** to the last green deploy (`bin/rollback.sh vX.Y.Z`). + - **Failover** to the healthy replicas (Caddy LB removes the bad + replica automatically; verify with `curl /api/health/ping`). + - **Disable** the broken connection(s) via `PUT /api/providers/{connectionId}` + with body `{ "isActive": false }` (per-connection toggle, safe by + default; repeat per key/account — see § 4.1). +6. **0:15** — Post the chosen mitigation in the channel. If the page + is still firing after 5 more minutes, escalate to the secondary. + +[chan]: TBD — set to your team's incident-chat channel (e.g. a Discord/Slack `#inc-*` channel); not provisioned by this repo. +[dash]: TBD — set to your Grafana/observability dashboard URL; not provisioned by this repo. + +--- + +## 4. Mitigation runbooks (per failure mode) + +### 4.1 Provider outage (single provider down) + +1. `PUT /api/providers/{connectionId}` with body `{ "isActive": false }` — + deactivates that connection; combo routing and account selection skip it + on the next request (`src/app/api/providers/[id]/route.ts`). There is no + single whole-provider kill switch — if the provider has more than one + key/account, repeat per connection, or let the automatic provider circuit + breaker trip on its own (`src/shared/utils/circuitBreaker.ts`, + `domain_circuit_breakers` table; see `docs/architecture/RESILIENCE_GUIDE.md`). +2. Verify p95 returns to budget within 5 min. +3. If all connections for a model are down, apply the same `isActive: false` + toggle to every connection offering that model — there is no separate + per-model disable endpoint. Combo routing's automatic Model Lockout + (`open-sse/services/accountFallback.ts`; see + `docs/architecture/RESILIENCE_GUIDE.md`) also skips a model that keeps + erroring, without manual action. +4. Update the status page (if one is configured — see § 5) with a banner if + the outage exceeds 15 min. + +### 4.2 Cluster-wide latency regression + +1. Check the most recent deploy (`/api/version` returns the SHA). +2. If p95 doubled vs the 7-day baseline, **roll back** to the prior + SHA via `bin/rollback.sh`. +3. If the regression is provider-side, see § 4.1. + +### 4.3 Auth layer broken (5xx on /v1/responses for all keys) + +1. Check the authz-inventory endpoint: + `curl https://api.omniroute.dev/api/settings/authz-inventory | jq`. + It returns a route-tier inventory (`tiers`, `bypassEnabled`, + `bypassPrefixes`, `spawnCapablePrefixes`, `cors` — see + `src/app/api/settings/authz-inventory/route.ts`); there is no + `policies_active` field. A non-200 response, or a `tiers` array that + fails to populate, means the settings/DB layer the auth pipeline reads + from is down — not just a single bad key. +2. If the endpoint itself errors or returns malformed data, restore the + settings store from the last good backup (`bin/restore-policies.sh `). +3. If the endpoint is healthy but requests still 5xx for every key, verify + `JWT_SECRET` / `API_KEY_SECRET` are set and unchanged for this deploy, + and that `isValidApiKey` (`src/sse/services/auth.ts`) can reach the DB. +4. Roll back if the cause is unclear. + +### 4.4 Data-layer incident (sqlite corruption, audit log gap) + +1. **Stop the cluster** (`docker compose -f docker-compose.prod.yml + stop`) — preventing further writes is more important than uptime. +2. Snapshot the data volume (`bin/snapshot-data.sh`). +3. Open a SEV-1; this is data-loss territory. Page the data-team. +4. Restore from the last verified backup (see `docs/BACKUP.md` once + published; currently the runbook is `bin/restore-data.sh `). + +### 4.5 Security incident (vulnerability disclosure) + +**Stop.** This is the `SECURITY.md` path, not this runbook. Page the +security on-call (`@security-team`); do not post details to +`#omniroute-ops`. + +--- + +## 5. Communication + +| Audience | Channel | Cadence | Owner | +|---|---|---|---| +| Engineering | `#inc-YYYY-MM-DD-slug` | Real-time | Incident commander | +| Status page | TBD — not provisioned by this repo | Every 30 min during SEV-1/2 | On-call | +| Customers (email) | TBD — set your announcement list/address | At SEV-1 start + resolution | Comms lead | +| Upstream providers | Direct contact | At SEV-1 start | Vendor mgmt | +| Postmortem | `docs/postmortem/YYYY-MM-DD-slug.md` | Within 5 business days | Incident commander | + +Postmortem template is at `docs/postmortem/TEMPLATE.md` (forthcoming; no +dedicated ADR covers it yet — once written, register it in +`docs/architecture/cluster-decisions.md` following this repo's 71-pillar/ADR +numbering convention, e.g. ADR-041 there). + +--- + +## 6. On-call rotation + +| Role | Primary | Secondary | Rotation | +|---|---|---|---| +| Engineering on-call | security-circle lead | @open-sse | Weekly, Mon 09:00 PDT | +| Security on-call | @security-team | — | Weekly | +| Data on-call | @db-team | — | Weekly | +| Comms lead | @comms | — | As needed | + +**Handoff**: every Monday 09:00 PDT, the outgoing on-call posts a +written handoff to the incoming in `#omniroute-ops-handoff` covering: +open SEV-3/4 items, scheduled maintenance windows, and any +in-flight mitigations. + +--- + +## 7. Postmortem expectations + +- **Blameless**. People did the best they could with the information + they had. Focus on systems, signals, and decision points. +- **Within 5 business days** of resolution. File via + `gh issue create --label postmortem --label SEV-1` (or `--label SEV-2`). +- **Action items** must be assigned, dated, and tracked in + `docs/TECH_DEBT.md` (P0 < 30 d, P1 < 90 d per that doc's SLA). +- **Mandatory attendees**: incident commander, on-call, any engineer + who touched the mitigation, and one person who was *not* involved + (fresh-eyes review). + +--- + +## 8. Review log + +| Date | Reviewer | Change | +|---|---|---| +| 2026-06-18 | security-circle lead | Initial runbook; severity ladder + 15-min checklist + 4.1–4.5 mitigation runbooks. Closes 71-pillar audit L61 (1/3 → 2/3). | +| 2026-07-18 | observability-circle | Corrected § 4.1/4.3 to the real provider-disable (`PUT /api/providers/{connectionId}`) and authz-inventory (`tiers`/`bypassEnabled`/`cors`, no `policies_active`) mechanisms; removed foreign branding and the nonexistent ADR-024/029 references. | +| 2026-07-18 (planned) | observability-circle | Wire on-call rotation into PagerDuty schedule; add the postmortem template. | diff --git a/docs/PERF_BUDGETS.md b/docs/PERF_BUDGETS.md new file mode 100644 index 00000000000..ca7af81f12e --- /dev/null +++ b/docs/PERF_BUDGETS.md @@ -0,0 +1,227 @@ +# Performance Budgets — OmniRoute (2026-06-18) + +**Status**: Authoritative. SLO targets that the 71-pillar audit (L13) +references for the `Perf > 2.00` gate. +**Methodology**: per-endpoint p50/p95/p99 latency budgets, plus a +top-level availability SLO. Budgets are derived from the 3-replica +Caddy + Redis topology (commit `038439fa7`); adjust on infra change. +**Enforcement**: none yet. § 6 sketches a `benches/perf-gate.k6.js` k6 +script that would assert the SLOs below, but it is a design reference, +not a committed file — no `bench/` or `benches/` directory exists in +this repo today. This doc is a target-setting reference only until a +CI gate is built as follow-up work. +**Re-evaluation cadence**: quarterly, or on any major infra change. + +--- + +## 1. Top-level SLOs + +| SLO | Target | Window | Page on breach | +|---|---|---|---| +| **Availability** (2xx or 4xx for /v1/* and /api/settings/*) | 99.9% | rolling 30 days | on-call P2 | +| **Error budget burn rate** (1xx normalized rate) | < 2x for 1h, < 6x for 5m | 1h / 5m windows | on-call P1 | +| **Aggregate p95 latency** (all /v1/*) | ≤ 1.5 s | rolling 5 min | on-call P2 | +| **Aggregate p99 latency** (all /v1/*) | ≤ 4.0 s | rolling 5 min | on-call P2 | + +**Error budget**: 30-day window = 43.2 minutes of unavailability at +99.9%. Burn rate > 2x is P2; > 6x is P1. + +--- + +## 2. Per-endpoint latency budgets + +All budgets measured **server-side** (Next.js Route Handler entry to +response start, or last byte for streaming). Stream endpoints are +measured to time-of-first-byte (TTFB) since the body is incremental. + +### 2.1 Inference endpoints (the hot path) + +| Endpoint | Method | p50 | p95 | p99 | Notes | +|---|---|---|---|---|---| +| `/v1/responses` (non-stream) | POST | 800 ms | 1.8 s | 3.5 s | Includes translator + provider roundtrip | +| `/v1/responses` (stream) | POST (TTFB) | 350 ms | 900 ms | 1.8 s | TTFB only; total duration unbounded | +| `/v1/relay/chat/completions` (non-stream) | POST | 1.0 s | 2.2 s | 4.0 s | Includes per-(token,IP) rate-limit check | +| `/v1/relay/chat/completions` (stream) | POST (TTFB) | 400 ms | 1.0 s | 2.0 s | | +| `/v1/embeddings` | POST | 300 ms | 700 ms | 1.4 s | Pure provider roundtrip; cheap | +| `/v1/rerank` | POST | 600 ms | 1.4 s | 2.8 s | | +| `/v1/moderations` | POST | 250 ms | 600 ms | 1.2 s | Lightweight classification | +| `/v1/audio/speech` | POST | 1.2 s | 3.0 s | 6.0 s | Audio synthesis is slow; budget reflects that | +| `/v1/audio/transcriptions` | POST | 2.0 s | 5.0 s | 10.0 s | STT is bounded by audio duration + model size | +| `/v1/images/generations` | POST | 4.0 s | 8.0 s | 15.0 s | Image gen is async-bound by provider | +| `/v1/videos/generations` | POST (TTFB) | 600 ms | 1.5 s | 3.0 s | Async; client polls `/v1/videos/{id}` | +| `/v1/music/generations` | POST | 3.0 s | 6.0 s | 12.0 s | | + +### 2.2 Files + batches + +| Endpoint | Method | p50 | p95 | p99 | Notes | +|---|---|---|---|---|---| +| `/v1/files` (GET) | GET | 80 ms | 200 ms | 400 ms | Cached list | +| `/v1/files` (POST upload) | POST | 500 ms | 1.2 s | 2.5 s | 25 MB cap; multipart parse | +| `/v1/files/{id}` (GET) | GET | 60 ms | 150 ms | 300 ms | | +| `/v1/files/{id}` (DELETE) | DELETE | 80 ms | 200 ms | 400 ms | | +| `/v1/files/{id}/content` (download) | GET | 100 ms | 300 ms | 600 ms | + per-MB throughput | +| `/v1/batches` (GET) | GET | 150 ms | 400 ms | 800 ms | | +| `/v1/batches` (POST create) | POST | 200 ms | 500 ms | 1.0 s | Validates input file then enqueues | +| `/v1/batches/{id}` (GET) | GET | 100 ms | 300 ms | 600 ms | | +| `/v1/batches/{id}` (DELETE) | DELETE | 100 ms | 300 ms | 600 ms | | +| `/v1/batches/delete-completed` (POST) | POST | 400 ms | 1.0 s | 2.0 s | Mass delete; n rows | + +### 2.3 Agents + +| Endpoint | Method | p50 | p95 | p99 | Notes | +|---|---|---|---|---|---| +| `/v1/agents/health` | GET | 1.5 s | 4.5 s | 5.0 s | 5s per-provider timeout cap; expect 3-provider total | +| `/v1/agents/credentials` | GET | 100 ms | 250 ms | 500 ms | Metadata only; values never returned | +| `/v1/agents/tasks` (GET list) | GET | 150 ms | 400 ms | 800 ms | | +| `/v1/agents/tasks` (POST create) | POST | 250 ms | 600 ms | 1.2 s | Just enqueues; doesn't run agent | +| `/v1/agents/tasks/{id}` (GET) | GET | 100 ms | 300 ms | 600 ms | | +| `/v1/agents/tasks/{id}` (DELETE) | DELETE | 150 ms | 400 ms | 800 ms | | + +### 2.4 Combos / me / providers + +| Endpoint | Method | p50 | p95 | p99 | +|---|---|---|---|---| +| `/v1/combos` | GET | 80 ms | 200 ms | 400 ms | +| `/v1/me/status` | GET | 60 ms | 150 ms | 300 ms | +| `/v1/providers/{provider}/models` | GET | 100 ms | 250 ms | 500 ms | + +### 2.5 Web / search + +| Endpoint | Method | p50 | p95 | p99 | Notes | +|---|---|---|---|---|---| +| `/v1/web/fetch` | POST | 1.5 s | 4.0 s | 8.0 s | 10s timeout cap; recurse depth 3 | +| `/v1/search` | POST | 800 ms | 2.0 s | 4.0 s | Provider search latency varies | + +### 2.6 VSCode-CLI shim (token-scoped) + +These are the legacy passthrough paths. Budgets are tighter because +they're called frequently by the VSCode-CLI extension in tight loops. + +| Endpoint | Method | p50 | p95 | p99 | +|---|---|---|---|---| +| `/v1/vscode/{token}/v1/chat/completions` | POST | 700 ms | 1.6 s | 3.0 s | +| `/v1/vscode/{token}/v1/models` | GET | 60 ms | 150 ms | 300 ms | +| `/v1/vscode/{token}/combos` | GET | 80 ms | 200 ms | 400 ms | +| `/v1/vscode/{token}/chat/completions` (legacy) | POST | 700 ms | 1.6 s | 3.0 s | +| `/v1/vscode/{token}/models` (legacy) | GET | 60 ms | 150 ms | 300 ms | +| `/v1/vscode/{token}/responses` | POST | 800 ms | 1.8 s | 3.5 s | + +### 2.7 Management / settings + +Management endpoints are operator-only and not part of the hot path. +Budgets are set conservatively; breaches don't page on-call but do +flag in the weekly perf review. + +| Endpoint group | p50 | p95 | p99 | +|---|---|---|---| +| `/api/settings/*` (GET) | 100 ms | 300 ms | 600 ms | +| `/api/settings/*` (POST/PATCH/DELETE) | 200 ms | 500 ms | 1.0 s | +| `/api/keys/*` (CRUD) | 150 ms | 400 ms | 800 ms | +| `/api/quota/*` (CRUD) | 150 ms | 400 ms | 800 ms | +| `/api/monitoring/health` (heavy) | 500 ms | 1.5 s | 3.0 s | + +### 2.8 Public probes + +| Endpoint | Method | p50 | p95 | p99 | +|---|---|---|---|---| +| `/api/health/ping` | GET | 5 ms | 20 ms | 50 ms | +| `/api/version` | GET | 5 ms | 20 ms | 50 ms | +| `/api/docs` | GET | 20 ms | 80 ms | 200 ms (HTML shell, no provider call) | + +--- + +## 3. Throughput targets + +| Tier | Per-replica RPS | Cluster RPS (3 replicas) | Notes | +|---|---|---|---| +| Inference (non-stream) | 50 RPS | 150 RPS | Bounded by provider quota + translator CPU | +| Inference (stream) | 25 concurrent streams | 75 streams | Bounded by Node event-loop + memory | +| Embeddings | 200 RPS | 600 RPS | Cheap | +| Files (upload) | 10 RPS | 30 RPS | Multipart parse + DB write | +| Files (download) | 100 RPS | 300 RPS | Static-content via Next.js | +| Combos / me / providers | 500 RPS | 1,500 RPS | Cached | +| WebSocket | 100 concurrent connections | 300 | Per-IP cap 5 | + +**Cluster ceiling** (all endpoints combined, sustained): ~1,000 RPS +before p95 latency begins to climb. Scale horizontally beyond that +by adding replicas; the Caddy LB is stateless. + +--- + +## 4. Resource budgets + +| Resource | Per-replica cap | Notes | +|---|---|---| +| RSS memory | 1.5 GB | Spikes during audio/video gen; expect brief 2 GB | +| Event-loop lag (p99) | 50 ms | Alert via `clinic doctor` regression | +| Heap retained | 800 MB | Old-gen GC tuning in `node --max-old-space-size` | +| File descriptors | 2,000 | `ulimit -n 4096` recommended at host | +| DB connections (sql.js) | 1 per replica | sql.js is in-process; no pool needed | +| Redis connections | 20 per replica | Pooled; idle reaped at 5 min | + +--- + +## 5. Cold-start budget + +Next.js App Router cold-start on a fresh container: + +| Phase | Budget | +|---|---| +| Container start → HTTP listening | ≤ 800 ms | +| First request TTFB (warm) | ≤ 200 ms | +| Translator registry bootstrap | ≤ 500 ms (one-time, first /v1/responses) | + +**Measurement script**: `bin/cold-start-bench.sh` (already in the repo +since v3.8.36; `bin/` is the canonical scripts dir). + +--- + +## 6. Regression gate (k6 reference, not yet implemented) + +The sketch below shows how a future `benches/perf-gate.k6.js` script +would assert the SLOs above. Nothing in this section is committed or +wired into CI today — it is a design reference for follow-up work, not +a running gate. + +```javascript +// benches/perf-gate.k6.js — pseudo-code; not yet committed +import http from 'k6/http'; +import { check, Trend } from 'k6'; + +const responsesTTFB = new Trend('v1_responses_ttfb', true); + +export const options = { + scenarios: { + smoke: { + executor: 'constant-vus', + vus: 10, + duration: '1m', + }, + }, + thresholds: { + 'http_req_duration{endpoint:v1_responses}': ['p(95)<1800', 'p(99)<3500'], + 'http_req_failed': ['rate<0.01'], + 'v1_responses_ttfb': ['p(95)<900'], + }, +}; + +export default function () { + const res = http.post(`${__ENV.BASE_URL}/api/v1/responses`, JSON.stringify({ + model: 'gpt-4o-mini', + input: 'ping', + }), { headers: { 'Authorization': `Bearer ${__ENV.API_KEY}` }}); + check(res, { 'status is 200': (r) => r.status === 200 }); + responsesTTFB.add(res.timings.waiting); +} +``` + +--- + +## 7. Review log + +| Date | Reviewer | Change | +|---|---|---| +| 2026-06-18 | security-circle lead | Initial per-endpoint budgets derived from 3-replica Caddy + Redis topology | +| 2026-07-18 | observability-circle | Clarified this doc ships zero enforcement today (no `bench/`/`benches/` dir, no CI gate) and fixed the stale "not yet committed" claim about `bin/cold-start-bench.sh` (present since v3.8.36). | +| 2026-07-18 (planned) | observability-circle | Wire `benches/perf-gate.k6.js` into CI; gate on p95 + p99 breach | +| 2026-09-18 (planned) | observability-circle | Quarterly review; adjust after real-traffic baseline data | diff --git a/docs/compression/COMPRESSION_GUIDE.md b/docs/compression/COMPRESSION_GUIDE.md index 326a042a1c1..314458f2874 100644 --- a/docs/compression/COMPRESSION_GUIDE.md +++ b/docs/compression/COMPRESSION_GUIDE.md @@ -93,6 +93,8 @@ RTK mode is optimized for verbose tool outputs that appear in coding-agent sessi TypeScript/Vite/Webpack builds, ESLint/Biome/Prettier, npm audit/installs, Docker logs, infra output, and generic shell output - Applies JSON filter packs from `open-sse/services/compression/engines/rtk/filters/` +- Imports RTK TOML schema v1 filters from project or global `filters.toml` files, with inline-test + validation and trust-gating for project files - Ships 49 built-in filters with inline verify samples - Removes ANSI control sequences, progress bars, repeated lines, and non-actionable noise - Preserves failures, errors, warnings, changed files, summaries, and the tail of long output diff --git a/docs/compression/RTK_COMPRESSION.md b/docs/compression/RTK_COMPRESSION.md index 4457cda3da1..6ac7f0d2d77 100644 --- a/docs/compression/RTK_COMPRESSION.md +++ b/docs/compression/RTK_COMPRESSION.md @@ -52,28 +52,59 @@ class is not enough. RTK loads filters in this order: -1. Project filters from `.rtk/filters.json`, only when trusted. -2. Global filters from `DATA_DIR/rtk/filters.json`. +1. Project filters from `.rtk/filters.toml` and `.rtk/filters.json`, only when trusted. +2. Global filters from `DATA_DIR/rtk/filters.toml` and `DATA_DIR/rtk/filters.json`. 3. Built-in filters from `open-sse/services/compression/engines/rtk/filters/`. +Within the same scope, RTK TOML schema v1 filters take precedence over OmniRoute JSON filters. TOML +`match_command` expressions are checked before command-type matching so an imported command-specific +filter can override a broader filter in that scope. Project scope still takes precedence over global +scope, regardless of file format. + Project filters are intentionally trust-gated because regex filters can change how tool output is shown to agents. A project filter file is accepted when one of these is true: - `rtkConfig.trustProjectFilters` is `true`. - `OMNIROUTE_RTK_TRUST_PROJECT_FILTERS=1` is set. -- `.rtk/trust.json` contains the SHA-256 hash of `.rtk/filters.json`. +- `.rtk/trust.json` contains the matching SHA-256 hash for the project filter file. Trust file example: ```json { - "filtersSha256": "0123456789abcdef..." + "filtersSha256": "0123456789abcdef...", + "filtersTomlSha256": "fedcba9876543210..." } ``` +The hashes are separate: `filtersSha256` trusts `.rtk/filters.json`, while `filtersTomlSha256` +trusts `.rtk/filters.toml`. Editing either file invalidates only its own trust entry. Global files +are administrator-installed and use the existing global-filter trust behavior. + Custom filters can be one filter object or an array of filter objects. Invalid custom filters are skipped and reported by `/api/context/rtk/filters` diagnostics. Invalid built-in filters fail fast. +## RTK TOML schema v1 compatibility + +OmniRoute can parse, validate, test, and install declarative filter files using RTK TOML schema v1. +The supported fields are `description`, `match_command`, `strip_ansi`, `filter_stderr`, +`strip_lines_matching`, `keep_lines_matching`, `replace`, `match_output`, `truncate_lines_at`, +`head_lines`, `tail_lines`, `max_lines`, `on_empty`, and `[[tests.]]` inline tests. +Unknown fields, invalid or unsafe regular expressions, simultaneous strip/keep rules, files over +1 MiB, and references to unknown filters are rejected. A file whose inline tests fail can be +validated for inspection but cannot be installed or loaded. Custom-file load failures remain +fail-open: the invalid file is skipped and the remaining filters continue to work. + +OmniRoute receives tool output after the client has already captured it, so `filter_stderr = true` +cannot change process capture. The field is accepted as a no-op and validation returns a warning. +This is intentionally described as **RTK TOML schema v1 compatibility**, not full compatibility +with the RTK executable, shell hooks, Rust command implementations, or its trust-store layout. + +The dashboard's advanced RTK view accepts pasted or uploaded TOML. Validation is read-only. +Installation writes `DATA_DIR/rtk/filters.toml` atomically with restrictive permissions and refreshes +the live filter catalog without a restart. Replacing an existing file requires explicit `overwrite` +confirmation and creates `DATA_DIR/rtk/filters.toml.bak` first. + ## Filter DSL Filters use the JSON schema described in [Compression Rules Format](./COMPRESSION_RULES_FORMAT.md). @@ -216,6 +247,7 @@ round-trips through the same store and survives a restart. | `/api/context/rtk/config` | GET | Read RTK config | | `/api/context/rtk/config` | PUT | Update RTK config | | `/api/context/rtk/filters` | GET | List filter catalog and load diagnostics | +| `/api/context/rtk/import` | POST | Validate or install RTK TOML schema v1 files | | `/api/context/rtk/test` | POST | Preview RTK compression for one text payload | | `/api/context/rtk/raw-output/[id]` | GET | Read retained redacted raw output | | `/api/compression/preview` | POST | Preview any compression mode | @@ -253,6 +285,18 @@ Compression preview payload: Management routes require dashboard management auth or the matching API-key policy. +RTK TOML validation payload: + +```json +{ + "action": "validate", + "content": "schema_version = 1\n\n[filters.my-tool]\nmatch_command = \"^my-tool\\\\b\"\nmax_lines = 20\n" +} +``` + +Use `"action": "install"` to install the validated file globally. Add `"overwrite": true` only +after reviewing and confirming replacement of an existing global file. + ## Raw Output Recovery RTK normally returns only compressed text. For debugging, `rawOutputRetention` can retain redacted diff --git a/docs/frameworks/MCP-SERVER.md b/docs/frameworks/MCP-SERVER.md index 1f953da93e0..73fcf367afc 100644 --- a/docs/frameworks/MCP-SERVER.md +++ b/docs/frameworks/MCP-SERVER.md @@ -6,13 +6,9 @@ lastUpdated: 2026-06-28 # OmniRoute MCP Server Documentation -> Model Context Protocol server with 94 tools across routing, cache, compression, memory, skills, proxy, pool, and context source operations. +> Model Context Protocol server with 104 tools across routing, cache, compression, memory, skills, proxy, pool, and context source operations. > -> Source of truth: `open-sse/mcp-server/schemas/tools.ts` (34 base) + `memoryTools.ts` (3) + `skillTools.ts` (4) + `agentSkillTools.ts` (3) + `poolTools.ts` (6) + `gamificationTools.ts` (8) + `pluginTools.ts` (8) + `notionTools.ts` (6) + `obsidianTools.ts` (22) = **94** (`TOTAL_MCP_TOOL_COUNT`). Tool registration and scope wiring lives in `open-sse/mcp-server/server.ts`. - -![MCP tool inventory (94 tools by category)](../diagrams/exported/mcp-tools-94.svg) - -> Source: [diagrams/mcp-tools-94.mmd](../diagrams/mcp-tools-94.mmd) (regenerate via `npm run docs:render-diagrams`). +> Source of truth: `open-sse/mcp-server/server.ts` computes **104 unique tools** with `countUniqueMcpTools()`: 42 canonical definitions (including the six CCR lifecycle tools and the agent-skills trio), plus memory (3), skills (4), GitHub skills (3), pool (6), gamification (8), plugins (8), Notion (6), Obsidian (22), and two RTK-only compression tools. ## Installation @@ -110,7 +106,7 @@ Cursor, Cline, and compatible MCP client setup. | `omniroute_cache_stats` | `read:cache` | Semantic cache, prompt-cache, and idempotency stats | | `omniroute_cache_flush` | `write:cache` | Flush cache globally or by signature/model | -## Compression Tools (5) +## Compression Tools (13) | Tool | Scopes | Description | | :---------------------------------- | :------------------ | :----------------------------------------------------------------------------------------------------------------------- | @@ -119,6 +115,20 @@ Cursor, Cline, and compatible MCP client setup. | `omniroute_set_compression_engine` | `write:compression` | Pick the active engine (off/caveman/rtk/stacked) and Caveman/RTK intensity | | `omniroute_list_compression_combos` | `read:compression` | List named compression combos and their engine pipelines | | `omniroute_compression_combo_stats` | `read:compression` | Analytics grouped by compression combo and engine | +| `omniroute_ccr_store` | `write:compression` | Store caller-isolated content in the bounded in-memory CCR store and return a marker plus `ccr://` reference | +| `omniroute_ccr_retrieve` | `read:compression` | Retrieve CCR content in full or with head, tail, lines, grep, and stats modes | +| `omniroute_ccr_inspect` | `read:compression` | Inspect caller-owned CCR metadata without returning content | +| `omniroute_ccr_list` | `read:compression` | List paginated metadata for caller-owned CCR blocks | +| `omniroute_ccr_delete` | `write:compression` | Delete a caller-owned CCR block | +| `omniroute_ccr_stats` | `read:compression` | Report caller-scoped memory usage, lifecycle counters, and store limits | +| `omniroute_rtk_discover` | `read:compression` | Discover recurring noise in opt-in RTK output samples | +| `omniroute_rtk_learn` | `read:compression` | Generate a reviewable RTK filter draft from opt-in samples | + +CCR entries are in-memory only and disappear on restart. Each block is limited to 2 MiB, each +principal to 16 MiB, and the global store to 64 MiB. Entries default to a 24-hour TTL (maximum +seven days). Full MCP retrieval is limited to 256 KiB; larger blocks remain available through the +ranged and grep modes. Storage, retrieval, listing, inspection, deletion, and stats are isolated by +the authenticated API-key principal. Audit records contain hashes and size metadata, never content. `omniroute_compression_status` reports MCP description compression separately under `analytics.mcpDescriptionCompression`. Those values are metadata-size estimates for MCP listable @@ -127,7 +137,7 @@ receipts and are marked with `source: "mcp_metadata_estimate"`. ### MCP Accessibility Tree Filter (v3.8.0) -Separate from the 5 compression tools above, OmniRoute includes a post-execution filter that +Separate from the compression tools above, OmniRoute includes a post-execution filter that compresses the **tool results** of MCP browser/accessibility tools before they are returned to the agent. This filter is not itself a tool — it runs transparently on any tool result that contains verbose accessibility-tree or browser-snapshot text (≥2000 chars). @@ -217,7 +227,7 @@ See [AGENT-SKILLS.md](./AGENT-SKILLS.md) for the full catalog and how external a ## Related Frameworks (v3.8.0) -The MCP tool inventory above (94 tools = 34 core + 3 memory + 4 skills + 3 agent-skills + 6 pool + 8 gamification + 8 plugins + 6 notion + 22 obsidian) is intentionally +The MCP tool inventory above (104 unique tools, computed by `countUniqueMcpTools()`) is intentionally scoped to runtime routing/cache/compression/memory/skills/proxy/context-source operations. Two adjacent frameworks ship alongside the MCP server in v3.8.0 and are documented separately: @@ -331,7 +341,7 @@ MCP tool, prompt, and resource registries can compress descriptions at registrat Description compression shrinks each tool's metadata; **tool-cardinality reduction** goes one step further by reducing _how many_ tools are announced at all. Advertising fewer tools in the `tools/list` manifest cuts the per-request token cost the client's model pays for the tool catalog ("layer 5" compression). The implementation is a pure, stateless filter in `open-sse/mcp-server/toolCardinality.ts` (`reduceToolManifest`), wired into the registration loop in `createMcpServer()` (`open-sse/mcp-server/server.ts`). -**Opt-in, off by default.** The filter only runs when at least one of two environment variables is set; with neither set, all 94 tools are announced unchanged. +**Opt-in, off by default.** The filter only runs when at least one of two environment variables is set; with neither set, all 104 tools are announced unchanged. | Variable | Mode | | :--------------- | :-------------------------------------------------------------------------------------- | diff --git a/docs/openapi.yaml b/docs/openapi.yaml index 994154d41d5..6c9241849b0 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -2201,6 +2201,36 @@ paths: "200": description: RTK filter catalog and diagnostics + /api/context/rtk/import: + post: + tags: [Compression] + summary: Validate or install an RTK TOML schema v1 filter file + security: + - ManagementSessionAuth: [] + requestBody: + required: true + content: + application/json: + schema: + type: object + required: [action, content] + additionalProperties: false + properties: + action: + type: string + enum: [validate, install] + content: + type: string + maxLength: 1048576 + overwrite: + type: boolean + description: Replace an existing global file and create a backup + responses: + "200": + description: Filter metadata, inline-test outcomes, warnings, and installation status + "400": + description: Invalid TOML, schema, regular expression, inline test, or install request + /api/context/rtk/test: post: tags: [Compression] diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 9b39ccee2ed..6527b4b3031 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -411,6 +411,7 @@ detection above). | `OMNIROUTE_API_KEY` | _(unset)_ | MCP/A2A modules | API key for internal MCP tool and A2A skill calls. | | `OMNIROUTE_API_KEY_ID` | _(unset)_ | `open-sse/mcp-server/audit.ts` | Key ID for MCP audit log attribution. | | `ROUTER_API_KEY` | _(unset)_ | Legacy | Legacy alias for `OMNIROUTE_API_KEY`. | +| `OMNIROUTE_ISSUE_AGENT_ENABLED` | `false` | `src/app/api/issue-agent/runs/route.ts` | Enables the offline/local Issue Agent recorded-triage endpoint. Leave disabled unless explicitly running local recorded-triage workflows. | | `OMNIROUTE_CONTEXT` | _(active context)_ | `bin/cli/program.mjs`, `bin/cli/api.mjs` | CLI remote-mode context/profile for `omniroute` commands; overrides the active context in the local contexts store. Equivalent to `--context `. | | `OMNIROUTE_MCP_ENFORCE_SCOPES` | `true` | `open-sse/mcp-server/server.ts` | Enforce scope-based access control on MCP tool calls. | | `OMNIROUTE_MCP_SCOPES` | _(all)_ | `open-sse/mcp-server/server.ts` | Comma-separated scopes: `admin`, `combos`, `health`, `models`, `routing`, `budget`, `metrics`, `pricing`, `memory`, `skills`. | diff --git a/docs/reference/PROVIDER_PLUGIN_MANIFEST.md b/docs/reference/PROVIDER_PLUGIN_MANIFEST.md index 9f91dc39eda..9db239b371d 100644 --- a/docs/reference/PROVIDER_PLUGIN_MANIFEST.md +++ b/docs/reference/PROVIDER_PLUGIN_MANIFEST.md @@ -21,6 +21,14 @@ OmniRoute advertises that URL to Bifrost and CLIProxyAPI via the `OMNIROUTE_PROVIDER_MANIFEST_URL` when the sidecar needs a public or container network URL instead of the local request origin. +## Refreshing the Manifest + +The HTTP endpoint returns `Cache-Control: public, max-age=60` and a strong +`ETag`. A sidecar should retain the last validated manifest and send its ETag +in `If-None-Match` when refreshing. A `304 Not Modified` response has no body; +the sidecar keeps its cached manifest. If no validated cached manifest exists, +the sidecar must issue an unconditional request instead of accepting a `304`. + ## Goal Move provider metadata toward a plugin contract so the hot request path can diff --git a/docs/sessions/20260714-issue-agent-executable-triage/00_SESSION_OVERVIEW.md b/docs/sessions/20260714-issue-agent-executable-triage/00_SESSION_OVERVIEW.md new file mode 100644 index 00000000000..378c4454e94 --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/00_SESSION_OVERVIEW.md @@ -0,0 +1,36 @@ +# Issue-Agent Executable Triage: Session Overview + +Machine status: `in_progress` +Updated at: `2026-07-14` +Issue: `https://github.com/diegosouzapw/OmniRoute/issues/5980` +PR: `https://github.com/diegosouzapw/OmniRoute/pull/7002` + +## Goal + +Deliver GitHub issue #5980 as a production issue-agent workflow. The workflow +must execute recorded GitHub triage through OmniRoute routing, persist a complete +audit trail, return an actionable result, and cover all terminal outcomes. + +## Current State + +| artifact_id | requirement | status | current evidence | next proof | +| ----------- | ---------------------------------------------------------------------- | -------------------------------- | ---------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | +| AC1 | configured provider/model/policy use normal chat routing | `implemented_pending_acceptance` | `623e0d541`, `fa2c1d7c6`; real-route test invokes the issue-agent route and mocks only provider HTTP | prove routing-policy semantics and terminal failure handling | +| AC2 | persist lifecycle, request, output, usage/cost/runtime, terminal error | `not_started` | audit JSONL currently records only pre-execution run context | lifecycle persistence tests | +| AC3 | return actionable triage result | `not_started` | route forwards raw completion body | result contract and integration test | +| AC4 | success, provider failure, timeout, budget stop | `not_started` | only success-route coverage exists | terminal-outcome test matrix | +| release | CI/review evidence | `in_progress` | route-validation and focused tests have prior passing evidence | rerun final gates on PR head | + +## Decisions + +| decision_id | decision | rationale | status | +| ----------- | -------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------- | ------------- | +| DEC-001 | Use the in-process `POST` export from `/api/v1/chat/completions` | preserves existing admission, initialization, guardrails, and provider routing | `implemented` | +| DEC-002 | Keep issue-agent execution opt-in with `OMNIROUTE_ISSUE_AGENT_ENABLED=true` | prevents unrequested autonomous execution | `implemented` | +| DEC-003 | Treat AC1 as incomplete until policy and error semantics are verified end-to-end | request construction alone does not prove the chat route consumes the policy or returns correct terminal state | `active` | + +## Traceability + +The canonical WBS is `03_DAG_WBS.md`; the canonical QA matrix is +`06_TESTING_STRATEGY.md`. Every status change must identify its commit SHA, +exact command, observed result, and PR head. diff --git a/docs/sessions/20260714-issue-agent-executable-triage/01_RESEARCH.md b/docs/sessions/20260714-issue-agent-executable-triage/01_RESEARCH.md new file mode 100644 index 00000000000..d0af543d875 --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/01_RESEARCH.md @@ -0,0 +1,28 @@ +# Issue-Agent Executable Triage: Research + +Machine status: `complete_for_current_phase` + +## In-Repository Findings + +| research_id | source | finding | consequence | +| ----------- | ------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- | +| RES-001 | `src/app/api/issue-agent/runs/route.ts` | the endpoint validates body, rejects unsupported mode/disabled execution, builds recorded context, writes audit JSONL, then delegates non-dry runs | execution behavior is centralized at the issue-agent route | +| RES-002 | `src/app/api/v1/chat/completions/route.ts` | standard chat entrypoint exports `POST` and owns the normal chat request path | AC1 must exercise this export rather than a fake internal seam | +| RES-003 | `src/lib/issueAgent/execution.ts` | provider and model are resolved into the chat request; policy is only encoded as `X-OmniRoute-Mode` | an implementation review must establish that this header is a consumed routing-policy contract | +| RES-004 | `src/lib/issueAgent/audit.ts` | audit persistence occurs before execution and writes run context/steps only | AC2 is unsatisfied: no transition, completion, usage/cost/runtime, or terminal-error record exists | +| RES-005 | `tests/unit/issue-agent-route-execution.test.ts` | live route test initializes isolated DB, calls the actual issue-agent `POST`, and mocks only `globalThis.fetch` at provider boundary | strong AC1 path evidence, but it verifies success only and does not prove policy consumption | + +## Validation Evidence + +| evidence_id | command | observed | scope | evidence_sha | +| ----------- | -------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------- | ----------------------------------------------------- | ------------ | +| EVD-001 | `bun test tests/unit/issue-agent-execution.test.ts tests/unit/issue-agent-route-execution.test.ts tests/unit/issue-agent-runs-route.test.ts` | prior focused run reported green | AC1 focused path | `fa2c1d7c6` | +| EVD-002 | `npm run check:route-validation:t06` | prior run reported pass | request route validation | `e6a63eb33` | +| EVD-003 | `npm run typecheck:core` | unresolved `omniglyph` declarations outside issue-agent paths | release gate blocked by pre-existing unrelated errors | pre-existing | + +## Research Conclusions + +The normal chat route is correctly selected as the AC1 integration seam. The +remaining design work must use a persisted run-lifecycle model rather than +extending the pre-execution JSONL row. No external API research was needed: +the implementation uses existing in-repository routes and provider adapters. diff --git a/docs/sessions/20260714-issue-agent-executable-triage/02_SPECIFICATIONS.md b/docs/sessions/20260714-issue-agent-executable-triage/02_SPECIFICATIONS.md new file mode 100644 index 00000000000..636b543a54d --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/02_SPECIFICATIONS.md @@ -0,0 +1,43 @@ +# Issue-Agent Executable Triage: Specifications + +Machine status: `in_progress` + +## Acceptance Contract + +| ac_id | requirement | acceptance evidence | status | +| ----- | --------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- | +| AC1 | Non-dry recorded triage executes through normal chat routing with selected provider, model, and policy | actual issue-agent route reaches chat `POST`; provider-boundary mock observes selected target; policy is proven consumed by routing | `implemented_pending_acceptance` | +| AC2 | Persist `accepted`, `running`, and terminal state plus sanitized request/prompt, model output, usage, cost, runtime, and terminal error | durable queryable record contains each field for success and failures | `pending` | +| AC3 | API returns a useful, structured triage result derived from model output | response has stable triage schema and is not a raw opaque provider payload | `pending` | +| AC4 | Tests cover success, provider/model failure, timeout, and budget stop | each outcome asserts HTTP response and persisted terminal record | `pending` | + +## API Contract (Target) + +| field | rule | +| ------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| `mode` | must be `recorded-triage` | +| execution selection | accepts configured `provider`, `model`, `routingPolicy`, and bounded `timeoutMs` | +| `runId` | stable execution identifier returned for every accepted run | +| result | includes structured triage decision/summary/actions and execution metadata | +| errors | return sanitized terminal error with explicit terminal status; never leak provider credentials or unredacted issue content | + +## Persistence Contract (Target) + +| field group | required values | +| -------------- | --------------------------------------------------------------------------------------------------------- | +| identity | run ID, issue URL/repository/number, mode, timestamps | +| lifecycle | `accepted`, `running`, `succeeded`, `failed`, `timed_out`, or `budget_stopped` with transition timestamps | +| input | redacted recorded context and rendered prompt fingerprint/content according to retention policy | +| routing | requested provider/model/policy and resolved execution target | +| output | sanitized model output and structured triage result | +| accounting | input/output/total tokens, cost, and runtime when available | +| terminal error | normalized code/message for failure, timeout, and budget stop | + +## Assumptions, Risks, Uncertainties + +| aru_id | type | statement | mitigation | status | +| ------- | ----------- | ---------------------------------------------------------------------------------------- | ------------------------------------------------------------------------- | ------ | +| ARU-001 | risk | `X-OmniRoute-Mode` may not be a consumed routing-policy input in the chat route | trace the policy contract and test an observable policy effect | `open` | +| ARU-002 | risk | current catch maps all thrown execution errors to HTTP 400 and does not persist them | introduce typed terminal outcomes and persistence before response mapping | `open` | +| ARU-003 | risk | current audit row is emitted before execution and cannot represent final execution state | replace/extend with append-only lifecycle records or durable run storage | `open` | +| ARU-004 | uncertainty | provider response metadata may differ by adapter | normalize accounting fields and preserve unknowns explicitly | `open` | diff --git a/docs/sessions/20260714-issue-agent-executable-triage/03_DAG_WBS.md b/docs/sessions/20260714-issue-agent-executable-triage/03_DAG_WBS.md new file mode 100644 index 00000000000..284a885c5c9 --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/03_DAG_WBS.md @@ -0,0 +1,25 @@ +# Issue-Agent Executable Triage: DAG and WBS + +Machine status: `in_progress` + +| id | phase | acceptance criterion | status | source paths | test paths | evidence_sha | depends_on | +| ------- | ----------- | ---------------------------------------------------------------- | -------- | ----------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------ | --------------------------- | ------------------------- | +| WBS-001 | contract | AC1-AC4 | complete | `src/app/api/issue-agent/runs/route.ts` | `tests/unit/issue-agent-runs-route.test.ts` | `a4378a26d` | - | +| WBS-002 | execution | AC1: execute through normal chat-completions routing/policy seam | pending | `src/app/api/issue-agent/runs/route.ts`; `src/app/api/v1/chat/completions/route.ts` | `tests/unit/issue-agent-runs-route.test.ts` | `e6a` (reconciled baseline) | WBS-001 | +| WBS-003 | persistence | AC2: persist lifecycle, input, output, usage, and terminal error | pending | `src/lib/issueAgent/*`; `src/app/api/issue-agent/runs/route.ts` | `tests/unit/issue-agent-audit.test.ts`; `tests/unit/issue-agent-runner.test.ts` | `e6a` (reconciled baseline) | WBS-002 | +| WBS-004 | result | AC3: return an actionable triage result from execution | pending | `src/lib/issueAgent/*`; `src/app/api/issue-agent/runs/route.ts` | `tests/unit/issue-agent-runner.test.ts`; `tests/unit/issue-agent-runs-route.test.ts` | `e6a` (reconciled baseline) | WBS-002, WBS-003 | +| WBS-005 | acceptance | AC4: cover success, provider failure, timeout, and budget stop | pending | `src/lib/issueAgent/*` | `tests/unit/issue-agent-*.test.ts` | `e6a` (reconciled baseline) | WBS-002, WBS-003, WBS-004 | +| WBS-006 | release | PR validation and maintainer review | pending | `.github/workflows/*` | CI checks | `a4378a26d` | WBS-005 | + +## Dependency Graph + +`WBS-001 -> WBS-002 -> WBS-003 -> WBS-004 -> WBS-005 -> WBS-006` + +`a4378a26d` is a prerequisite validation repair: it validates the issue-agent request body through the shared route validator and passes `npm run check:route-validation:t06` (535 routes). It does not satisfy AC1-AC4. + +## Machine Evidence Contract + +Every WBS item must maintain: `id`, `acceptance_criterion`, `status`, `source_paths`, `test_paths`, `command`, `expected`, `observed`, `evidence_sha`, `updated_at`, and `pr_url`. + +PR: `https://github.com/diegosouzapw/OmniRoute/pull/7002` +Issue: `https://github.com/diegosouzapw/OmniRoute/issues/5980` diff --git a/docs/sessions/20260714-issue-agent-executable-triage/04_IMPLEMENTATION_STRATEGY.md b/docs/sessions/20260714-issue-agent-executable-triage/04_IMPLEMENTATION_STRATEGY.md new file mode 100644 index 00000000000..1c591749739 --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/04_IMPLEMENTATION_STRATEGY.md @@ -0,0 +1,37 @@ +# Issue-Agent Executable Triage: Implementation Strategy + +Machine status: `in_progress` + +## Phase Plan + +| phase | work package | dependency | exit evidence | status | +| ----- | ----------------------------------------------------------- | ----------------- | ----------------------------------------------------------------------- | ------------- | +| P1 | verify/finish routing-policy contract and failure semantics | existing AC1 seam | actual chat route test proves policy consumption and non-2xx mapping | `in_progress` | +| P2 | introduce durable execution lifecycle persistence | P1 | records transitions, request/prompt, output, accounting, terminal error | `pending` | +| P3 | normalize actionable triage result | P2 | stable API result schema derived from completion | `pending` | +| P4 | implement terminal outcome controls | P2 | provider failure, timeout, budget stop transition tests | `pending` | +| P5 | release validation and PR review | P1-P4 | focused tests, route gate, relevant typecheck/CI evidence | `pending` | + +## Architecture + +1. Keep `src/app/api/issue-agent/runs/route.ts` as the API adapter: validation, + feature gate, and response formatting only. +2. Keep the standard chat `POST` as the routing boundary; do not add a parallel + provider invocation path. +3. Extract lifecycle persistence and result normalization into focused + `src/lib/issueAgent/` modules. Do not overload the existing pre-execution audit + writer with unrelated transport behavior. +4. Use typed execution outcomes so provider failure, abort/timeout, and budget + termination are distinguishable before HTTP mapping and persistence. +5. Add tests from the actual route down to a mocked external provider boundary; + use unit tests for pure normalization and lifecycle state transitions. + +## Quality Controls + +| control | command or review | threshold | +| ------------------ | ----------------------------------------------------- | ---------------------------------------------------------- | +| route contract | `npm run check:route-validation:t06` | pass | +| AC1 route behavior | focused `bun test` issue-agent route/execution suites | policy and provider/model assertions pass | +| AC2-AC4 | lifecycle/result/terminal-outcome suites | all required states persist and API matches | +| static safety | `npm run typecheck:core` | distinguish new failures from existing `omniglyph` blocker | +| patch integrity | `git diff --check origin/main...HEAD` | pass | diff --git a/docs/sessions/20260714-issue-agent-executable-triage/05_KNOWN_ISSUES.md b/docs/sessions/20260714-issue-agent-executable-triage/05_KNOWN_ISSUES.md new file mode 100644 index 00000000000..6c3f0fbe0e3 --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/05_KNOWN_ISSUES.md @@ -0,0 +1,21 @@ +# Issue-Agent Executable Triage: Known Issues + +Machine status: `open` + +| issue_id | severity | status | evidence | impact | resolution owner | +| -------- | -------- | ------ | ------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- | ------------------------- | +| KI-001 | P1 | `open` | `execution.ts` places `routingPolicy` in `X-OmniRoute-Mode`; the researched chat route has no observed consumer in the AC1 path | AC1 does not yet prove configured routing policy affects routing | AC1 implementation/review | +| KI-002 | P1 | `open` | issue-agent route catches execution errors and returns `{ error }` with HTTP 400 after writing only pre-execution audit | provider failure, timeout, and budget stop lack correct terminal semantics and persistence | AC2/AC4 implementation | +| KI-003 | P1 | `open` | `audit.ts` serializes only run context/steps before execution | AC2 fields for lifecycle, prompt, output, token/cost/runtime, and error are missing | AC2 implementation | +| KI-004 | P1 | `open` | API returns raw `completion.body` | AC3 has no stable actionable triage result contract | AC3 implementation | +| KI-005 | P2 | `open` | `npm run typecheck:core` has unresolved `omniglyph` declarations in `open-sse/services/compression/*` | full typecheck cannot be used as issue-agent completion evidence until separately resolved or excluded with provenance | release validation | + +## Resolved/Verified + +| issue_id | status | evidence | +| -------- | ---------- | ----------------------------------------------------------------------------------------------------------------------------- | +| KI-R001 | `verified` | `fa2c1d7c6` adds an isolated test that invokes the actual issue-agent route and mocks only provider HTTP for the success path | +| KI-R002 | `verified` | `e6a63eb33` applies shared request-body validation to the issue-agent route; prior route-validation gate passed | + +No workaround in this document changes the acceptance contract. Open P1 items +block declaring AC1-AC4 complete. diff --git a/docs/sessions/20260714-issue-agent-executable-triage/06_TESTING_STRATEGY.md b/docs/sessions/20260714-issue-agent-executable-triage/06_TESTING_STRATEGY.md new file mode 100644 index 00000000000..006908ef71a --- /dev/null +++ b/docs/sessions/20260714-issue-agent-executable-triage/06_TESTING_STRATEGY.md @@ -0,0 +1,25 @@ +# Issue-Agent Executable Triage: Testing Strategy + +Machine status: `in_progress` + +## QA Matrix + +| qa_id | AC | scenario | command | expected | observed | status | evidence_sha | +| ------ | ------------ | ------------------------------------------------------------------- | ------------------------------------------------------------------------------------------ | ----------------------------------------------------- | --------------------------------------- | ------------- | ------------ | +| QA-001 | prerequisite | request schema validation | `npm run check:route-validation:t06` | all routes pass | 535 routes scanned; pass | pass | `a4378a26d` | +| QA-002 | AC1 | selected provider/model/policy reaches normal chat-completions seam | `bun test tests/unit/issue-agent-runs-route.test.ts` | captured request uses configured routing inputs | not implemented | pending | `e6a` | +| QA-003 | AC2 | run lifecycle persists input, output, usage, terminal error | `bun test tests/unit/issue-agent-audit.test.ts tests/unit/issue-agent-runner.test.ts` | durable records for every terminal state | not implemented | pending | `e6a` | +| QA-004 | AC3 | successful execution returns actionable triage output | `bun test tests/unit/issue-agent-runner.test.ts tests/unit/issue-agent-runs-route.test.ts` | output derives from routed execution, not placeholder | not implemented | pending | `e6a` | +| QA-005 | AC4 | provider/model failure | `bun test tests/unit/issue-agent-runner.test.ts` | failed lifecycle and sanitized error persisted | missing coverage | pending | `e6a` | +| QA-006 | AC4 | timeout | `bun test tests/unit/issue-agent-runner.test.ts` | timed-out lifecycle and terminal error persisted | missing coverage | pending | `e6a` | +| QA-007 | AC4 | budget stop | `bun test tests/unit/issue-agent-runner.test.ts` | budget stop is explicit and persisted | missing coverage | pending | `e6a` | +| QA-008 | release | core type safety | `npm run typecheck:core` | pass | pending rerun after dependency recovery | pending | `e6a` | +| QA-009 | release | whitespace integrity | `git diff --check origin/main...HEAD` | no errors | passed before remote rewrite | pass/reverify | `a4378a26d` | + +## Test Rules + +Tests must mock only the external provider boundary. AC1 must exercise the in-process `POST` export from `src/app/api/v1/chat/completions/route.ts` so admission, policy, translator initialization, and routing remain in the execution path. Each terminal outcome asserts both API behavior and persisted audit state. + +## Evidence Requirements + +Before a WBS item is marked complete, record the exact command output, commit SHA, test identifiers, and whether the test environment had a lockfile-compatible dependency set. The current recovered environment has incomplete dependencies due to `npm ci` disk exhaustion; no pending test may be reported as passing until rerun. diff --git a/electron/package-lock.json b/electron/package-lock.json index b5fda4c818d..65430313322 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -1,18 +1,18 @@ { "name": "omniroute-desktop", - "version": "3.8.46", + "version": "3.8.49", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "omniroute-desktop", - "version": "3.8.46", + "version": "3.8.49", "license": "MIT", "dependencies": { "electron-updater": "^6.8.9" }, "devDependencies": { - "electron": "^43.1.0", + "electron": "^43.1.1", "electron-builder": "^26.15.3" }, "engines": { @@ -1367,9 +1367,9 @@ } }, "node_modules/electron": { - "version": "43.1.0", - "resolved": "https://registry.npmjs.org/electron/-/electron-43.1.0.tgz", - "integrity": "sha512-DPfxpQLd4NL3BJ8DBxYAfmLUKKesF5Rx9dQx5FyczAP8bhOPScjHE48GArVeXu68LlAainuwkmQTQvdZwpIIAQ==", + "version": "43.1.1", + "resolved": "https://registry.npmjs.org/electron/-/electron-43.1.1.tgz", + "integrity": "sha512-I5c5vfuVvaXpWx3IZdwvXgxQW44+e7OP1wXGVQkogLeSFSkUZ6sLCcWV05AdEcs65AO5tAIJJwbp7ixw+LdarA==", "dev": true, "license": "MIT", "dependencies": { diff --git a/electron/package.json b/electron/package.json index b3b5f1a268e..fcae7078db2 100644 --- a/electron/package.json +++ b/electron/package.json @@ -28,7 +28,7 @@ "electron-updater": "^6.8.9" }, "devDependencies": { - "electron": "^43.1.0", + "electron": "^43.1.1", "electron-builder": "^26.15.3" }, "overrides": { diff --git a/next.config.mjs b/next.config.mjs index b3f7887bb5c..fa46e3adbd7 100644 --- a/next.config.mjs +++ b/next.config.mjs @@ -9,8 +9,8 @@ const distDir = process.env.NEXT_DIST_DIR || ".build/next"; const projectRoot = dirname(fileURLToPath(import.meta.url)); const scriptSrc = process.env.NODE_ENV === "development" - ? "script-src 'self' 'unsafe-inline' 'unsafe-eval' blob:" - : "script-src 'self' 'unsafe-inline' 'unsafe-eval' blob:"; + ? "script-src 'self' 'unsafe-inline' 'unsafe-eval' blob: https://static.cloudflareinsights.com" + : "script-src 'self' 'unsafe-inline' 'unsafe-eval' blob: https://static.cloudflareinsights.com"; const contentSecurityPolicy = [ "default-src 'self'", "base-uri 'self'", diff --git a/open-sse/config/agyModels.ts b/open-sse/config/agyModels.ts index d836888b7ff..919b97e9573 100644 --- a/open-sse/config/agyModels.ts +++ b/open-sse/config/agyModels.ts @@ -19,7 +19,7 @@ export const AGY_PUBLIC_MODELS = Object.freeze([ { id: "claude-opus-4-6-thinking", name: "Claude Opus 4.6 (Thinking)", - contextLength: 200000, + contextLength: 1048576, maxOutputTokens: 65536, supportsReasoning: true, supportsVision: true, @@ -28,7 +28,7 @@ export const AGY_PUBLIC_MODELS = Object.freeze([ { id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6 (Thinking)", - contextLength: 200000, + contextLength: 1048576, maxOutputTokens: 65536, supportsReasoning: true, supportsVision: true, diff --git a/open-sse/config/antigravityModelAliases.ts b/open-sse/config/antigravityModelAliases.ts index a88ab804c02..3feebf495b1 100644 --- a/open-sse/config/antigravityModelAliases.ts +++ b/open-sse/config/antigravityModelAliases.ts @@ -7,7 +7,7 @@ export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([ { id: "claude-sonnet-5", name: "Claude Sonnet 5 (Thinking)", - contextLength: 200000, + contextLength: 1048576, maxOutputTokens: 65536, supportsReasoning: true, supportsVision: true, @@ -16,7 +16,7 @@ export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([ { id: "claude-opus-4-6-thinking", name: "Claude Opus 4.6 (Thinking)", - contextLength: 200000, + contextLength: 1048576, maxOutputTokens: 65536, supportsReasoning: true, supportsVision: true, @@ -25,7 +25,7 @@ export const ANTIGRAVITY_PUBLIC_MODELS = Object.freeze([ { id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6 (Thinking)", - contextLength: 200000, + contextLength: 1048576, maxOutputTokens: 65536, supportsReasoning: true, supportsVision: true, diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index 6b807affb19..f1725e012e5 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -285,6 +285,7 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "nscale", modelId: "meta-llama/Llama-4-Scout-17B-16E-Instruct", displayName: "meta-llama/Llama-4-Scout-17B-16E-Instruct", monthlyTokens: 0, creditTokens: 5000000, freeType: "one-time-initial", poolKey: "nscale", tos: "caution" }, { provider: "nscale", modelId: "meta-llama/Llama-3.3-70B-Instruct", displayName: "meta-llama/Llama-3.3-70B-Instruct", monthlyTokens: 0, creditTokens: 5000000, freeType: "one-time-initial", poolKey: "nscale", tos: "caution" }, { provider: "nvidia", modelId: "z-ai/glm-5.1", displayName: "GLM 5.1", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, + { provider: "nvidia", modelId: "z-ai/glm-5.2", displayName: "GLM 5.2", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "minimaxai/minimax-m2.7", displayName: "MiniMax M2.7", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "google/gemma-4-31b-it", displayName: "Gemma 4 31B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "mistralai/mistral-small-4-119b-2603", displayName: "Mistral Small 4 2603", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, diff --git a/open-sse/config/imageRegistry.ts b/open-sse/config/imageRegistry.ts index 109d494b038..45803f41229 100644 --- a/open-sse/config/imageRegistry.ts +++ b/open-sse/config/imageRegistry.ts @@ -9,11 +9,16 @@ import { LMARENA_DIRECT_IMAGE_MODELS } from "./providers/registry/lmarena/direct import { SEGMIND_IMAGE_PROVIDER } from "./providers/registry/segmind/imageModels.ts"; import { KIE_IMAGE_MODELS } from "./providers/registry/kie/imageModels.ts"; import { FREEPIK_IMAGE_PROVIDER } from "./providers/registry/freepik/index.ts"; +import { STABILITY_AI_IMAGE_MODELS } from "./providers/registry/stability-ai/imageModels.ts"; +import { GEMINI_IMAGEN_PROVIDER } from "./providers/registry/gemini/imageModels.ts"; interface ImageModelEntry { id: string; name: string; inputModalities?: string[]; + // See STABILITY_AI_IMAGE_MODELS for why this exists: some models accept "text" + // but mechanically require an image regardless. + imageRequired?: boolean; description?: string; isMarket?: boolean; } @@ -38,6 +43,7 @@ interface ImageModelAliasEntry { name: string; listInCatalog: boolean; inputModalities?: string[]; + imageRequired?: boolean; description?: string; } @@ -126,6 +132,13 @@ function findImageModelConfig(providerId, modelId) { return provider.models.find((model) => model.id === modelId) || null; } +// Kept out of getImageModelEntry() (which sits at the complexity-ratchet cap) — an +// alias can override imageRequired directly, else it falls back to its target +// model's own flag. Consumers coerce the result with Boolean(), so no `?? false`. +function resolveAliasImageRequired(alias, modelConfig) { + return alias.imageRequired ?? modelConfig?.imageRequired; +} + export const IMAGE_PROVIDERS: Record = { openai: { id: "openai", @@ -280,6 +293,10 @@ export const IMAGE_PROVIDERS: Record = { supportedSizes: ["1024x1024"], }, + // Google AI Studio Imagen family — dedicated :predict endpoint, not generateContent. + // See providers/registry/gemini/imageModels.ts for the full rationale. + gemini: GEMINI_IMAGEN_PROVIDER, + //Curruntly no models serving nebius: { id: "nebius", @@ -472,32 +489,7 @@ export const IMAGE_PROVIDERS: Record = { authType: "apikey", authHeader: "bearer", format: "stability-ai", - models: [ - { id: "stable-image-ultra", name: "Stable Image Ultra" }, - { id: "stable-image-core", name: "Stable Image Core" }, - { id: "sd3.5-large-turbo", name: "sd3.5-large-turbo" }, - { id: "sd3.5-large", name: "sd3.5-large" }, - { id: "sd3.5-medium", name: "sd3.5-medium" }, - { id: "sd3.5-flash", name: "sd3.5-flash" }, - { id: "erase", name: "Erase", inputModalities: ["image"] }, - { id: "inpaint", name: "Inpaint", inputModalities: ["text", "image"] }, - { id: "outpaint", name: "Outpaint", inputModalities: ["text", "image"] }, - { id: "remove-background", name: "Remove Background", inputModalities: ["image"] }, - { id: "search-and-replace", name: "Search and Replace", inputModalities: ["text", "image"] }, - { id: "search-and-recolor", name: "Search and Recolor", inputModalities: ["text", "image"] }, - { - id: "replace-background-and-relight", - name: "Replace Background and Relight", - inputModalities: ["text", "image"], - }, - { id: "creative", name: "Creative Upscale", inputModalities: ["text", "image"] }, - { id: "fast", name: "Fast Upscale", inputModalities: ["image"] }, - { id: "conservative", name: "Conservative Upscale", inputModalities: ["image"] }, - { id: "sketch", name: "Sketch Control", inputModalities: ["text", "image"] }, - { id: "structure", name: "Structure Control", inputModalities: ["text", "image"] }, - { id: "style", name: "Style Control", inputModalities: ["text", "image"] }, - { id: "style-transfer", name: "Style Transfer", inputModalities: ["text", "image"] }, - ], + models: STABILITY_AI_IMAGE_MODELS, supportedSizes: ["1024x1024", "1024x1280", "1280x1024"], }, @@ -620,7 +612,9 @@ export const IMAGE_PROVIDERS: Record = { // beyond this seed list. huggingface: { id: "huggingface", - baseUrl: "https://api-inference.huggingface.co/models", + // HF retired api-inference.huggingface.co; text-to-image now routes through + // router.huggingface.co with the hf-inference provider pinned in the path. + baseUrl: "https://router.huggingface.co/hf-inference/models", authType: "apikey", authHeader: "bearer", format: "huggingface-image", @@ -766,6 +760,7 @@ export function getImageModelEntry(modelStr) { provider: alias.provider, model: alias.model, inputModalities: alias.inputModalities || modelConfig?.inputModalities || ["text"], + imageRequired: resolveAliasImageRequired(alias, modelConfig), description: alias.description || modelConfig?.description || undefined, }; } @@ -780,6 +775,18 @@ export function getImageModelEntry(modelStr) { provider, model, inputModalities: modelConfig.inputModalities || ["text"], + imageRequired: modelConfig.imageRequired, description: modelConfig.description || undefined, }; } + +/** + * An image input is only MANDATORY for edit-only models — those whose modalities + * are `["image"]` with no `"text"`. Models listing both `["text", "image"]` accept + * an image but can also run pure text-to-image, so they must NOT be gated on an + * image input (that gate previously blocked 41 dual-modality t2i models). + */ +export function modalitiesRequireImageInput(inputModalities) { + const list = Array.isArray(inputModalities) ? inputModalities : ["text"]; + return list.includes("image") && !list.includes("text"); +} diff --git a/open-sse/config/nvidiaHostedModels.snapshot.json b/open-sse/config/nvidiaHostedModels.snapshot.json new file mode 100644 index 00000000000..29bb66e0ad5 --- /dev/null +++ b/open-sse/config/nvidiaHostedModels.snapshot.json @@ -0,0 +1,18 @@ +[ + "deepseek-ai/deepseek-v4-pro", + "google/gemma-4-31b-it", + "minimaxai/minimax-m2.7", + "mistralai/devstral-2-123b-instruct-2512", + "mistralai/mistral-large-3-675b-instruct-2512", + "mistralai/mistral-small-4-119b-2603", + "nvidia/nemotron-3-super-120b-a12b", + "openai/gpt-oss-120b", + "openai/gpt-oss-20b", + "poolside/laguna-xs-2.1", + "qwen/qwen3.5-122b-a10b", + "qwen/qwen3.5-397b-a17b", + "stepfun-ai/step-3.5-flash", + "thinkingmachines/inkling", + "z-ai/glm-5.1", + "z-ai/glm-5.2" +] diff --git a/open-sse/config/providers/index.ts b/open-sse/config/providers/index.ts index ecdd7a82b5c..286ed60fbdd 100644 --- a/open-sse/config/providers/index.ts +++ b/open-sse/config/providers/index.ts @@ -123,6 +123,7 @@ import { gitlawbProvider } from "./registry/gitlawb/index.ts"; import { liquidProvider } from "./registry/liquid/index.ts"; import { deepinfraProvider } from "./registry/deepinfra/index.ts"; import { agyProvider } from "./registry/agy/index.ts"; +import { agnesProvider } from "./registry/agnes/index.ts"; import { udioProvider } from "./registry/udio/index.ts"; import { longcatProvider } from "./registry/longcat/index.ts"; import { vertex_partnerProvider } from "./registry/vertex/partner/index.ts"; @@ -318,6 +319,7 @@ export const REGISTRY: Record = { liquid: liquidProvider, deepinfra: deepinfraProvider, agy: agyProvider, + agnes: agnesProvider, udio: udioProvider, longcat: longcatProvider, "vertex-partner": vertex_partnerProvider, diff --git a/open-sse/config/providers/registry/agnes/index.ts b/open-sse/config/providers/registry/agnes/index.ts new file mode 100644 index 00000000000..c3120a562c0 --- /dev/null +++ b/open-sse/config/providers/registry/agnes/index.ts @@ -0,0 +1,26 @@ +import type { RegistryEntry } from "../../shared.ts"; +import { buildOpenAiCompatibleRegistryEntry } from "../../shared.ts"; + +export const agnesProvider: RegistryEntry = buildOpenAiCompatibleRegistryEntry({ + id: "agnes", + baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions", + models: [ + { + id: "agnes-2.0-flash", + name: "Agnes 2.0 Flash", + contextLength: 524288, + maxOutputTokens: 65536, + supportsReasoning: true, + supportsVision: true, + toolCalling: true, + interleavedField: "reasoning_content", + }, + { + id: "agnes-1.5-flash", + name: "Agnes 1.5 Flash", + contextLength: 262144, + maxOutputTokens: 65536, + supportsVision: true, + }, + ], +}); diff --git a/open-sse/config/providers/registry/auggie/index.ts b/open-sse/config/providers/registry/auggie/index.ts index ee346426a0e..1415c627a7a 100644 --- a/open-sse/config/providers/registry/auggie/index.ts +++ b/open-sse/config/providers/registry/auggie/index.ts @@ -3,6 +3,8 @@ import type { RegistryEntry } from "../../shared.ts"; // Augment / Auggie CLI — local no-auth provider. The executor spawns the // user's local `auggie` binary (auth handled entirely by `auggie login`); // OmniRoute never stores credentials for this connection. +// +// Model IDs sourced from `auggie model list` on auggie v0.32.0. export const auggieProvider: RegistryEntry = { id: "auggie", alias: "aug", @@ -13,26 +15,38 @@ export const auggieProvider: RegistryEntry = { authHeader: "none", defaultContextLength: 200000, models: [ - // Claude - { id: "claude-sonnet-4.6", name: "Claude Sonnet 4.6", contextLength: 200000 }, - { - id: "claude-sonnet-4.6-thinking", - name: "Claude Sonnet 4.6 Thinking", - contextLength: 200000, - }, - { id: "claude-opus-4.6", name: "Claude Opus 4.6", contextLength: 200000 }, - { id: "claude-haiku-4.5", name: "Claude Haiku 4.5", contextLength: 200000 }, - // Gemini - { id: "gemini-3.1-pro", name: "Gemini 3.1 Pro", contextLength: 1000000 }, - { id: "gemini-3.0-flash", name: "Gemini 3 Flash", contextLength: 1000000 }, - // GPT-5.x - { id: "gpt-5.5-high", name: "GPT-5.5 High", contextLength: 200000 }, - { id: "gpt-5.5-medium", name: "GPT-5.5 Medium", contextLength: 200000 }, - { id: "gpt-5.4-high", name: "GPT-5.4 High", contextLength: 200000 }, - { id: "gpt-5.4-medium", name: "GPT-5.4 Medium", contextLength: 200000 }, - // Kimi + // ── Anthropic Claude ──────────────────────────────────────────────── + { id: "sonnet4.6", name: "Sonnet 4.6", contextLength: 200000 }, + { id: "fable-5", name: "Claude Fable 5", contextLength: 200000 }, + { id: "haiku4.5", name: "Haiku 4.5", contextLength: 200000 }, + { id: "sonnet4.5", name: "Sonnet 4.5", contextLength: 200000 }, + { id: "sonnet4.6-500k", name: "Sonnet 4.6 (500K)", contextLength: 500000 }, + { id: "sonnet5-high", name: "Claude Sonnet 5", contextLength: 200000 }, + { id: "sonnet5-500k", name: "Claude Sonnet 5 (500K)", contextLength: 500000 }, + { id: "opus4.5", name: "Opus 4.5", contextLength: 200000 }, + { id: "opus4.6", name: "Opus 4.6", contextLength: 200000 }, + { id: "opus4.6-500k", name: "Opus 4.6 (500K)", contextLength: 500000 }, + { id: "opus4.7", name: "Opus 4.7", contextLength: 200000 }, + { id: "opus4.7-500k", name: "Opus 4.7 (500K)", contextLength: 500000 }, + { id: "opus4.8", name: "Opus 4.8", contextLength: 200000 }, + // ── Gemini ────────────────────────────────────────────────────────── + { id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro", contextLength: 1000000 }, + // ── OpenAI GPT ────────────────────────────────────────────────────── + { id: "gpt5", name: "GPT-5", contextLength: 200000 }, + { id: "gpt5.1", name: "GPT-5.1", contextLength: 200000 }, + { id: "gpt5.2", name: "GPT-5.2", contextLength: 200000 }, + { id: "gpt5.4", name: "GPT-5.4", contextLength: 200000 }, + { id: "gpt5.4-mini", name: "GPT-5.4 Mini", contextLength: 200000 }, + { id: "gpt5.5", name: "GPT-5.5", contextLength: 200000 }, + { id: "gpt5.6-luna", name: "GPT-5.6 Luna", contextLength: 200000 }, + { id: "gpt5.6-sol", name: "GPT-5.6 Sol", contextLength: 200000 }, + { id: "gpt5.6-terra", name: "GPT-5.6 Terra", contextLength: 200000 }, + // ── Others ────────────────────────────────────────────────────────── + { id: "glm-5.2", name: "GLM 5.2", contextLength: 200000 }, { id: "kimi-k2.6", name: "Kimi K2.6", contextLength: 131000 }, - // Prism (Augment's in-house model) - { id: "prism", name: "Augment Prism", contextLength: 200000 }, + { id: "kimi-k2.7", name: "Kimi K2.7 Code", contextLength: 131000 }, + // ── Augment Prism (composite routers) ─────────────────────────────── + { id: "prism-a", name: "Prism (Claude + Gemini)", contextLength: 200000 }, + { id: "prism-b", name: "Prism (GPT + Kimi)", contextLength: 200000 }, ], }; diff --git a/open-sse/config/providers/registry/claude/index.ts b/open-sse/config/providers/registry/claude/index.ts index 19b2c5a5096..d142872de84 100644 --- a/open-sse/config/providers/registry/claude/index.ts +++ b/open-sse/config/providers/registry/claude/index.ts @@ -81,7 +81,7 @@ export const claudeProvider: RegistryEntry = { id: "claude-sonnet-4-6", name: "Claude 4.6 Sonnet", supportsXHighEffort: false, - contextLength: 200000, + contextLength: 1000000, maxOutputTokens: 64000, }, { diff --git a/open-sse/config/providers/registry/gemini/imageModels.ts b/open-sse/config/providers/registry/gemini/imageModels.ts new file mode 100644 index 00000000000..bc7002da013 --- /dev/null +++ b/open-sse/config/providers/registry/gemini/imageModels.ts @@ -0,0 +1,32 @@ +/** + * Google AI Studio (Gemini API) Imagen family image-generation provider entry. + * + * Uses the dedicated `:predict` endpoint (handled by format "google-imagen"), NOT + * generateContent — so only imagen-* models belong here; gemini flash-image / + * nano-banana route through /v1/chat/completions instead. The models are also + * surfaced live via ListModels; this seed makes them addressable on + * /v1/images/generations. Note: Imagen requires a billing-enabled Google project — + * free-tier keys get 403 / quota 0. The handler builds `{baseUrl}/{model}:predict`. + * + * Extracted out of imageRegistry.ts (which sits right at the 800-line file-size + * cap) so the catalog lives in its own semantic family module, following the same + * pattern as `providers/registry/stability-ai/imageModels.ts` and + * `providers/registry/segmind/imageModels.ts`. Co-located with the existing + * `gemini/index.ts` chat-provider entry — same provider id, different + * modality/consumer (chat registry vs image registry), mirroring the + * `kie/index.ts` + `kie/imageModels.ts` split. + */ +export const GEMINI_IMAGEN_PROVIDER = { + id: "gemini", + alias: "gemini", + baseUrl: "https://generativelanguage.googleapis.com/v1beta/models", + authType: "apikey", + authHeader: "x-goog-api-key", + format: "google-imagen", + models: [ + { id: "imagen-4.0-generate-001", name: "Imagen 4" }, + { id: "imagen-4.0-ultra-generate-001", name: "Imagen 4 Ultra" }, + { id: "imagen-4.0-fast-generate-001", name: "Imagen 4 Fast" }, + ], + supportedSizes: ["1024x1024", "1792x1024", "1024x1792"], +}; diff --git a/open-sse/config/providers/registry/opencode/go/index.ts b/open-sse/config/providers/registry/opencode/go/index.ts index c8e575b185b..e3dd4b4d80a 100644 --- a/open-sse/config/providers/registry/opencode/go/index.ts +++ b/open-sse/config/providers/registry/opencode/go/index.ts @@ -17,14 +17,22 @@ export const opencode_goProvider: RegistryEntry = { // glm-5.2 is now advertised and Kimi chat traffic must route through // `kimi-k2.7-code` (the live API rejects the plain `kimi-k2.7` alias for // `/chat/completions`, even though the docs config example uses it). - { id: "glm-5.2", name: "GLM-5.2" }, + // GLM-5.2 — base model + effort-tier aliases (#6922). + // OpencodeExecutor rewrites the alias to the canonical id and injects + // reasoning_effort, mirroring the deepseek-v4-pro-* pattern. + { id: "glm-5.2", name: "GLM-5.2", supportsReasoning: true }, + { id: "glm-5.2-high", name: "GLM-5.2 (high effort)", supportsReasoning: true }, + { id: "glm-5.2-max", name: "GLM-5.2 (max effort)", supportsReasoning: true }, { id: "glm-5.1", name: "GLM-5.1" }, { id: "glm-5", name: "GLM-5" }, { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, { id: "kimi-k2.6", name: "Kimi K2.6" }, { id: "kimi-k2.5", name: "Kimi K2.5" }, - { id: "mimo-v2.5-pro", name: "MiMo-V2.5-Pro" }, - { id: "mimo-v2.5", name: "MiMo-V2.5" }, + // MiMo-V2.5 — base model + effort-tier aliases (#6922). + { id: "mimo-v2.5-pro", name: "MiMo-V2.5-Pro", supportsReasoning: true }, + { id: "mimo-v2.5", name: "MiMo-V2.5", supportsReasoning: true }, + { id: "mimo-v2.5-high", name: "MiMo-V2.5 (high effort)", supportsReasoning: true }, + { id: "mimo-v2.5-max", name: "MiMo-V2.5 (max effort)", supportsReasoning: true }, // #3110: MiniMax M3 via OpenCode Go tier { id: "minimax-m3", diff --git a/open-sse/config/providers/registry/stability-ai/imageModels.ts b/open-sse/config/providers/registry/stability-ai/imageModels.ts new file mode 100644 index 00000000000..4a5b20623e9 --- /dev/null +++ b/open-sse/config/providers/registry/stability-ai/imageModels.ts @@ -0,0 +1,76 @@ +/** + * Stability AI image-generation model catalog. + * + * Extracted out of imageRegistry.ts (which sits right at the 800-line file-size + * cap) so the catalog lives in its own semantic family module, following the same + * pattern as `providers/registry/kie/imageModels.ts` and + * `providers/registry/segmind/imageModels.ts`. See `imageRegistry.ts`'s + * `stability-ai` entry for baseUrl/auth/format wiring. + * + * `imageRequired: true` marks the dedicated edit/control/upscale endpoints + * (STABILITY_EDIT_ENDPOINTS in open-sse/handlers/imageGeneration.ts) that accept a + * text prompt but mechanically require an input image regardless — + * modalitiesRequireImageInput() alone can't tell them apart from flexible + * dual-modality generation models (BFL Kontext, Together, NVIDIA, LMArena, + * NanoGPT), which correctly allow pure text-to-image. + */ + +export interface StabilityImageModelEntry { + id: string; + name: string; + inputModalities?: string[]; + imageRequired?: boolean; +} + +export const STABILITY_AI_IMAGE_MODELS: StabilityImageModelEntry[] = [ + { id: "stable-image-ultra", name: "Stable Image Ultra" }, + { id: "stable-image-core", name: "Stable Image Core" }, + { id: "sd3.5-large-turbo", name: "sd3.5-large-turbo" }, + { id: "sd3.5-large", name: "sd3.5-large" }, + { id: "sd3.5-medium", name: "sd3.5-medium" }, + { id: "sd3.5-flash", name: "sd3.5-flash" }, + { id: "erase", name: "Erase", inputModalities: ["image"] }, + { id: "inpaint", name: "Inpaint", inputModalities: ["text", "image"], imageRequired: true }, + { id: "outpaint", name: "Outpaint", inputModalities: ["text", "image"], imageRequired: true }, + { id: "remove-background", name: "Remove Background", inputModalities: ["image"] }, + { + id: "search-and-replace", + name: "Search and Replace", + inputModalities: ["text", "image"], + imageRequired: true, + }, + { + id: "search-and-recolor", + name: "Search and Recolor", + inputModalities: ["text", "image"], + imageRequired: true, + }, + { + id: "replace-background-and-relight", + name: "Replace Background and Relight", + inputModalities: ["text", "image"], + imageRequired: true, + }, + { + id: "creative", + name: "Creative Upscale", + inputModalities: ["text", "image"], + imageRequired: true, + }, + { id: "fast", name: "Fast Upscale", inputModalities: ["image"] }, + { id: "conservative", name: "Conservative Upscale", inputModalities: ["image"] }, + { id: "sketch", name: "Sketch Control", inputModalities: ["text", "image"], imageRequired: true }, + { + id: "structure", + name: "Structure Control", + inputModalities: ["text", "image"], + imageRequired: true, + }, + { id: "style", name: "Style Control", inputModalities: ["text", "image"], imageRequired: true }, + { + id: "style-transfer", + name: "Style Transfer", + inputModalities: ["text", "image"], + imageRequired: true, + }, +]; diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index e3fc66b378b..c3d4e4fbe8b 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -62,6 +62,10 @@ import { markCreditsExhausted, type SafeAntigravityLog, } from "./antigravity/executeAttempt.ts"; +import { + handleAntigravityFallbackChainError, + handleAntigravityFallback400, +} from "./antigravity/proFallbackChain.ts"; import { generateAntigravityRequestId, getAntigravityEnvelopeUserAgent, @@ -365,7 +369,18 @@ function sanitizeAntigravityGeminiRequest( const geminiTools = buildGeminiTools(request.tools); if (geminiTools) { clean.tools = geminiTools; - clean.toolConfig = { functionCallingConfig: { mode: "VALIDATED" } }; + // #6914: Preserve includeServerSideToolInvocations from the raw request's + // toolConfig when present (set by transformRequest when tools exist). The + // sanitize whitelist would otherwise rebuild toolConfig without it. + const rawToolConfig = asRecord(request.toolConfig); + const rawFnConfig = asRecord(rawToolConfig?.functionCallingConfig); + const includeServerSide = rawFnConfig?.includeServerSideToolInvocations === true; + clean.toolConfig = { + functionCallingConfig: { + mode: "VALIDATED", + ...(includeServerSide ? { includeServerSideToolInvocations: true } : {}), + }, + }; } else if (asRecord(request.toolConfig)) { clean.toolConfig = request.toolConfig; } @@ -641,7 +656,7 @@ export class AntigravityExecutor extends BaseExecutor { safetySettings: getAntigravitySafetySettings(normalizedRequest?.safetySettings), toolConfig: Array.isArray(normalizedRequest?.tools) && normalizedRequest.tools.length > 0 - ? { functionCallingConfig: { mode: "VALIDATED" } } + ? { functionCallingConfig: { mode: "VALIDATED", includeServerSideToolInvocations: true } } : normalizedRequest?.toolConfig, }; @@ -1015,7 +1030,28 @@ export class AntigravityExecutor extends BaseExecutor { let firstResult: Awaited> | null = null; for (let i = 0; i < chain.length; i++) { const candidate = chain[i]; - const result = await this.executeOnce(input, candidate); + let result: Awaited>; + try { + result = await this.executeOnce(input, candidate); + } catch (error) { + const outcome = handleAntigravityFallbackChainError( + input, + error, + candidate, + i, + chain, + firstResult, + resolvedUpstreamId + ); + switch (outcome.action) { + case "throw": + throw outcome.error; + case "return": + return outcome.result; + default: + continue; + } + } // Success (or any non-400) on a candidate → return immediately. if (result.response.status !== HTTP_STATUS.BAD_REQUEST) { @@ -1023,23 +1059,18 @@ export class AntigravityExecutor extends BaseExecutor { } // Remember the FIRST 400 so the exhausted-chain case surfaces the original error. - if (i === 0) firstResult = result; - - const isLast = i === chain.length - 1; - if (!isLast) { - input.log?.debug?.( - "AG_PRO_FALLBACK", - `400 on "${candidate}" — retrying with next Pro candidate "${chain[i + 1]}"` - ); - continue; - } - - // Chain exhausted: surface the FIRST candidate's sanitized 400. - input.log?.warn?.( - "AG_PRO_FALLBACK", - `Pro fallback chain exhausted (all ${chain.length} candidates 400'd) for "${resolvedUpstreamId}"` + if (!firstResult) firstResult = result; + + const outcome400 = handleAntigravityFallback400( + input, + result, + firstResult, + candidate, + i, + chain, + resolvedUpstreamId ); - return firstResult ?? result; + if (outcome400.action === "return") return outcome400.result; } // Unreachable (loop always returns), but keeps the type checker happy. diff --git a/open-sse/executors/antigravity/proFallbackChain.ts b/open-sse/executors/antigravity/proFallbackChain.ts new file mode 100644 index 00000000000..ec5b710c1f8 --- /dev/null +++ b/open-sse/executors/antigravity/proFallbackChain.ts @@ -0,0 +1,104 @@ +// Pure Pro-family fallback-chain decision helpers for the Antigravity executor (#7290): +// decide what execute()'s per-candidate loop does after executeOnce() throws or +// returns a 400, without depending on executor instance state (no `this`). +// Extracted from antigravity.ts (file-size cap) -- mirrors the existing +// antigravity/sseCollect.ts submodule pattern. +import type { ExecuteInput } from "../base.ts"; + +/** Shape of one execute()/executeOnce() result (kept local to avoid importing the class). */ +export type AntigravityExecuteResult = { + response: Response; + url: string; + headers: Record; + transformedBody: unknown; +}; + +/** True for an aborted request (caller disconnect) — never retried across candidates. */ +export function isAntigravityAbortError(input: ExecuteInput, error: unknown): boolean { + return Boolean( + input.signal?.aborted || + (error instanceof DOMException && error.name === "AbortError") || + (error instanceof Error && error.name === "AbortError") + ); +} + +export type AntigravityFallbackChainErrorOutcome = + | { action: "throw"; error: unknown } + | { action: "return"; result: AntigravityExecuteResult } + | { action: "continue" }; + +/** + * Decide what execute()'s Pro-fallback loop does after executeOnce() THROWS for one + * candidate: propagate an abort immediately, retry the next candidate, surface the + * first 400 if the chain is exhausted, or throw a chain-exhausted error. + */ +export function handleAntigravityFallbackChainError( + input: ExecuteInput, + error: unknown, + candidate: string, + i: number, + chain: readonly string[], + firstResult: AntigravityExecuteResult | null, + resolvedUpstreamId: string +): AntigravityFallbackChainErrorOutcome { + // Abort signal (user disconnect) — propagate immediately, do not retry. + if (isAntigravityAbortError(input, error)) { + return { action: "throw", error }; + } + if (i < chain.length - 1) { + input.log?.debug?.( + "AG_PRO_FALLBACK", + `Exception on "${candidate}" (${error instanceof Error ? error.message : String(error)}) -- retrying with next Pro candidate "${chain[i + 1]}"` + ); + return { action: "continue" }; + } + // Last candidate also threw -- return original 400 if available, otherwise throw. + if (firstResult) { + input.log?.warn?.( + "AG_PRO_FALLBACK", + `Pro fallback chain exhausted (last candidate threw, but first candidate returned 400) for "${resolvedUpstreamId}". Returning original 400.` + ); + return { action: "return", result: firstResult }; + } + return { + action: "throw", + error: new Error( + `Pro fallback chain exhausted (all ${chain.length} candidates failed). Last error: ${error instanceof Error ? error.message : String(error)}` + ), + }; +} + +export type AntigravityFallback400Outcome = + | { action: "return"; result: AntigravityExecuteResult } + | { action: "continue" }; + +/** + * Decide what execute()'s Pro-fallback loop does after one candidate returns a 400: + * retry the next candidate, or (chain exhausted) surface the first candidate's + * sanitized 400. + */ +export function handleAntigravityFallback400( + input: ExecuteInput, + result: AntigravityExecuteResult, + firstResult: AntigravityExecuteResult | null, + candidate: string, + i: number, + chain: readonly string[], + resolvedUpstreamId: string +): AntigravityFallback400Outcome { + const isLast = i === chain.length - 1; + if (!isLast) { + input.log?.debug?.( + "AG_PRO_FALLBACK", + `400 on "${candidate}" — retrying with next Pro candidate "${chain[i + 1]}"` + ); + return { action: "continue" }; + } + + // Chain exhausted: surface the FIRST candidate's sanitized 400. + input.log?.warn?.( + "AG_PRO_FALLBACK", + `Pro fallback chain exhausted (all ${chain.length} candidates 400'd) for "${resolvedUpstreamId}"` + ); + return { action: "return", result: firstResult ?? result }; +} diff --git a/open-sse/executors/antigravity/sseCollect.ts b/open-sse/executors/antigravity/sseCollect.ts index d42bab72e96..7c6e7957871 100644 --- a/open-sse/executors/antigravity/sseCollect.ts +++ b/open-sse/executors/antigravity/sseCollect.ts @@ -96,9 +96,23 @@ export function processAntigravitySSEPayload( collected.textContent += part.text; } } + // Native Gemini function calls. Non-streaming responses (and some + // streaming ones) carry the tool call as `part.functionCall` rather than + // the textual `[Tool call: ...]` markdown. Without this, a tool-only + // response produced empty content and a 502 Provider error (#7037). + if (part.functionCall && typeof part.functionCall.name === "string") { + addAntigravityTextualToolCall(collected, { + name: part.functionCall.name, + args: part.functionCall.args ?? {}, + }); + } } } - if (candidate?.finishReason) { + // Preserve a tool-call finish reason: once a native `part.functionCall` + // (or textual tool call) has populated `toolCalls`, the candidate's own + // finish reason (often STOP) must not clobber it (#7037 — a tool-only + // response would otherwise report STOP and lose its tool-call signal). + if (candidate?.finishReason && collected.toolCalls.length === 0) { collected.finishReason = normalizeOpenAICompatibleFinishReasonString( String(candidate.finishReason).toLowerCase() ); diff --git a/open-sse/executors/auggie.ts b/open-sse/executors/auggie.ts index b4034cbb2da..f3c98c8d8f9 100644 --- a/open-sse/executors/auggie.ts +++ b/open-sse/executors/auggie.ts @@ -38,8 +38,114 @@ const AUGGIE_URL = "auggie://cli/stdio"; // untrusted-input sink. We only ever pass a model that is declared in the // registry entry — this closes flag-smuggling (a `model` starting with "-" would // otherwise be parsed by auggie as an option) and unknown-model passthrough. +// +// The static registry (shipped with the code) is checked first. On first use +// the executor also spawns `auggie model list` at runtime and merges any IDs it +// finds — this lets the allowlist stay current when auggie adds or renames +// models without a code update. const AUGGIE_MODEL_ALLOWLIST: ReadonlySet = new Set(auggieProvider.models.map((m) => m.id)); const DEFAULT_AUGGIE_MODEL = auggieProvider.models[0]?.id ?? "claude-sonnet-4.6"; +// ─── Model alias map (backward compat for saved combos) ───────────────────── +// Old model IDs from before the v0.32.0 registry update; each maps to the +// equivalent v0.32.0 ID so existing combos continue to work after the rename. +const AUGGIE_MODEL_ALIASES: ReadonlyMap = new Map([ + // Claude + ["claude-sonnet-4.6", "sonnet4.6"], + ["claude-sonnet-4.6-thinking", "sonnet4.6"], + ["claude-opus-4.6", "opus4.6"], + ["claude-haiku-4.5", "haiku4.5"], + // Gemini + ["gemini-3.1-pro", "gemini-3.1-pro-preview"], + ["gemini-3.0-flash", "gemini-3.1-pro-preview"], + // GPT-5.x (high/medium split was synthetic — v0.32.0 has a single ID per version) + ["gpt-5.5-high", "gpt5.5"], + ["gpt-5.5-medium", "gpt5.5"], + ["gpt-5.4-high", "gpt5.4"], + ["gpt-5.4-medium", "gpt5.4"], +]); + +/** + * Live model cache populated by `initAuggieModels()`. + * - `null` = not yet attempted + * - `Set` = successfully fetched IDs (possibly empty) + */ +let liveModelSet: Set | null = null; + +/** + * Spawn `auggie model list`, parse `[model-id]` entries, and merge them into + * the live allowlist so the executor accepts models auggie recognises even + * when the static registry has not been updated yet. + * + * Safe to call repeatedly: only the first call spawns the process; subsequent + * calls are a no-op (including after a failed fetch — `liveModelSet` is set to + * an empty set so we don't retry every request). + */ +export async function initAuggieModels( + signal?: AbortSignal | null, + timeoutMs = 8000 +): Promise { + if (liveModelSet !== null) return; + let bin: string; + try { + bin = resolveAuggieBin(); + } catch { + liveModelSet = new Set(); + return; + } + const child = spawn(bin, ["model", "list"], { + env: process.env, + stdio: ["ignore", "pipe", "pipe"], + shell: false, + windowsHide: true, + }); + const fragments: string[] = []; + child.stdout.on("data", (d: Buffer) => fragments.push(d.toString("utf8"))); + let settled = false; + const settle = (result: Set) => { + if (settled) return; + settled = true; + liveModelSet = result; + }; + const timer = setTimeout(() => { + if (!child.killed) child.kill("SIGKILL"); + settle(new Set()); + }, timeoutMs); + const onAbort = () => { + if (!child.killed) child.kill("SIGKILL"); + clearTimeout(timer); + settle(new Set()); + }; + if (signal) { + if (signal.aborted) { + clearTimeout(timer); + settle(new Set()); + return; + } + signal.addEventListener("abort", onAbort, { once: true }); + } + try { + const code = await new Promise((resolve, reject) => { + child.on("close", resolve); + child.on("error", (e: Error) => reject(e)); + }); + clearTimeout(timer); + signal?.removeEventListener("abort", onAbort); + if (code !== 0) { + settle(new Set()); + return; + } + const ids = new Set(); + for (const line of fragments.join("").split("\n")) { + const m = line.match(/\[([^\]]+)\]/); + if (m) ids.add(m[1]); + } + settle(ids.size > 0 ? ids : new Set()); + } catch { + clearTimeout(timer); + settle(new Set()); + signal?.removeEventListener("abort", onAbort); + } +} type AuggieModelResolution = { ok: true; model: string } | { ok: false; error: string }; @@ -47,6 +153,9 @@ type AuggieModelResolution = { ok: true; model: string } | { ok: false; error: s * Validate + resolve the requested model against the registry allowlist. * Rejects flag-smuggling (leading "-") and any id not declared in the registry. * An empty/absent model resolves to the registry's first (default) model. + * + * Note: `initAuggieModels()` must be called at least once before this function + * sees live-discovered models (the executor's `execute()` does this). */ export function resolveAuggieModel(model: unknown): AuggieModelResolution { const requested = typeof model === "string" ? model.trim() : ""; @@ -57,15 +166,20 @@ export function resolveAuggieModel(model: unknown): AuggieModelResolution { error: `Invalid Auggie model "${requested}": model must not start with "-".`, }; } - if (!AUGGIE_MODEL_ALLOWLIST.has(requested)) { - return { - ok: false, - error: `Unknown Auggie model "${requested}". Supported models: ${[ - ...AUGGIE_MODEL_ALLOWLIST, - ].join(", ")}.`, - }; - } - return { ok: true, model: requested }; + // Backward-compat alias: resolve old model IDs → v0.32.0 equivalents. + // This lets saved combos referencing the old names keep working. + const requestedAlias = AUGGIE_MODEL_ALIASES.get(requested); + if (requestedAlias) return { ok: true, model: requestedAlias }; + // Static registry — always authoritative for the shipped set. + if (AUGGIE_MODEL_ALLOWLIST.has(requested)) return { ok: true, model: requested }; + // Live-discovered models (if loaded) extend the static list. + if (liveModelSet?.has(requested)) return { ok: true, model: requested }; + const known = [...AUGGIE_MODEL_ALLOWLIST]; + if (liveModelSet) known.push(...liveModelSet); + return { + ok: false, + error: `Unknown Auggie model "${requested}". Supported models: ${known.join(", ")}.`, + }; } /** @@ -151,6 +265,7 @@ export function buildAuggiePrompt(messages: OpenAIMsg[]): string { } } if (!text.trim()) continue; + if (role === "system") { lines.push(`[System]\n${text}`); } else if (role === "assistant") { @@ -252,7 +367,6 @@ export class AuggieExecutor extends BaseExecutor { ): Promise | null> { return null; } - async execute({ model, body, stream, signal, log }: ExecuteInput): Promise<{ response: Response; url: string; @@ -265,6 +379,9 @@ export class AuggieExecutor extends BaseExecutor { const auggieBin = resolveAuggieBin(); const wantsStream = stream !== false; + // On first execution, try to discover model IDs the local auggie recognises. + // Best-effort: missing/inactive CLI falls through to the static list. + await initAuggieModels(signal); // Argument-injection defense: never forward an unvalidated model into the argv. const modelResolution = resolveAuggieModel(model); if (!modelResolution.ok) { @@ -601,3 +718,13 @@ function buildAuggieSseError(message: string): Response { }, }); } + +// ─── Test helpers ────────────────────────────────────────────────────────── + +/** + * Reset the live model cache for testing. + * Not exported from the package index. + */ +export function __resetAuggieModels(): void { + liveModelSet = null; +} diff --git a/open-sse/executors/base/reasoningEffort.ts b/open-sse/executors/base/reasoningEffort.ts index dcabf6ef260..dac59232856 100644 --- a/open-sse/executors/base/reasoningEffort.ts +++ b/open-sse/executors/base/reasoningEffort.ts @@ -56,6 +56,86 @@ export function supportsMaxEffortForProvider(provider: string, model: string): b return isClaude || isOpencodeGoDeepSeek || isOllamaCloud; } +// ── Effort carrier helpers (#7044) ────────────────────────────────────────── +// OmniRoute carries the requested effort on up to three shapes: +// 1. top-level `reasoning_effort` — OpenAI / OmniRoute-internal +// 2. `reasoning.effort` — OpenAI Responses shape +// 3. `output_config.effort` — Anthropic Messages native (Claude Code / Claude passthrough) +// Carrier (3) was previously invisible to this sanitizer, so a native Claude request +// carrying `output_config.effort: "xhigh"` reached providers that don't accept xhigh +// (e.g. claude-sonnet-4-6, supportsXHighEffort=false) unchanged → HTTP 400 (#7044). +interface EffortCarriers { + reasoning: Record | null; + outputConfig: Record | null; + hasTopLevelReasoningEffort: boolean; + hasReasoningEffort: boolean; + hasOutputConfigEffort: boolean; + effort: unknown; +} + +function readEffortCarriers(b: Record): EffortCarriers { + const reasoning = + b.reasoning && typeof b.reasoning === "object" && !Array.isArray(b.reasoning) + ? (b.reasoning as Record) + : null; + const outputConfig = + b.output_config && typeof b.output_config === "object" && !Array.isArray(b.output_config) + ? (b.output_config as Record) + : null; + const hasTopLevelReasoningEffort = Object.prototype.hasOwnProperty.call(b, "reasoning_effort"); + const hasReasoningEffort = !!( + reasoning && Object.prototype.hasOwnProperty.call(reasoning, "effort") + ); + const hasOutputConfigEffort = !!( + outputConfig && Object.prototype.hasOwnProperty.call(outputConfig, "effort") + ); + const effort = b.reasoning_effort ?? reasoning?.effort ?? outputConfig?.effort; + return { + reasoning, + outputConfig, + hasTopLevelReasoningEffort, + hasReasoningEffort, + hasOutputConfigEffort, + effort, + }; +} + +/** Write a normalized effort value back to every carrier that was present. */ +function writeEffortValue( + b: Record, + value: string, + c: EffortCarriers +): Record { + const next: Record = { ...b }; + if (c.hasTopLevelReasoningEffort) next.reasoning_effort = value; + if (c.hasReasoningEffort && c.reasoning) next.reasoning = { ...c.reasoning, effort: value }; + if (c.hasOutputConfigEffort && c.outputConfig) + next.output_config = { ...c.outputConfig, effort: value }; + return next; +} + +/** Strip the effort field from every carrier that was present. */ +function stripEffortValue( + b: Record, + c: EffortCarriers +): Record { + const next: Record = { ...b }; + if (c.hasTopLevelReasoningEffort) delete next.reasoning_effort; + if (c.hasReasoningEffort && c.reasoning) { + const r: Record = { ...c.reasoning }; + delete r.effort; + if (Object.keys(r).length === 0) delete next.reasoning; + else next.reasoning = r; + } + if (c.hasOutputConfigEffort && c.outputConfig) { + const oc: Record = { ...c.outputConfig }; + delete oc.effort; + if (Object.keys(oc).length === 0) delete next.output_config; + else next.output_config = oc; + } + return next; +} + export function sanitizeReasoningEffortForProvider( body: unknown, provider: string, @@ -64,14 +144,9 @@ export function sanitizeReasoningEffortForProvider( ): unknown { if (!body || typeof body !== "object" || Array.isArray(body)) return body; const b = body as Record; - const reasoning = - b.reasoning && typeof b.reasoning === "object" && !Array.isArray(b.reasoning) - ? (b.reasoning as Record) - : null; - const hasTopLevelReasoningEffort = Object.prototype.hasOwnProperty.call(b, "reasoning_effort"); - const effort = b.reasoning_effort ?? reasoning?.effort; - if (effort === undefined) return body; - const effortStr = typeof effort === "string" ? effort.toLowerCase() : ""; + const c = readEffortCarriers(b); + if (c.effort === undefined) return body; + const effortStr = typeof c.effort === "string" ? c.effort.toLowerCase() : ""; const modelStr = model || ""; const githubOptIn = @@ -84,15 +159,7 @@ export function sanitizeReasoningEffortForProvider( "REASONING_SANITIZE", `${provider}/${modelStr}: removed unsupported reasoning_effort` ); - const next: Record = { ...b }; - delete next.reasoning_effort; - if (reasoning) { - const r = { ...reasoning }; - delete r.effort; - if (Object.keys(r).length === 0) delete next.reasoning; - else next.reasoning = r; - } - return next; + return stripEffortValue(b, c); } // Native DeepSeek (api.deepseek.com) — V4 thinking mode accepts reasoning_effort @@ -111,10 +178,7 @@ export function sanitizeReasoningEffortForProvider( "REASONING_SANITIZE", `deepseek/${modelStr}: normalized reasoning_effort ${effortStr} → ${mapped}` ); - const next: Record = { ...b }; - if (hasTopLevelReasoningEffort) next.reasoning_effort = mapped; - if (reasoning) next.reasoning = { ...reasoning, effort: mapped }; - return next; + return writeEffortValue(b, mapped, c); } return body; } @@ -131,14 +195,7 @@ export function sanitizeReasoningEffortForProvider( "REASONING_SANITIZE", `${provider}/${modelStr}: normalized reasoning_effort max → xhigh` ); - const next: Record = { ...b }; - if (hasTopLevelReasoningEffort) { - next.reasoning_effort = "xhigh"; - } - if (reasoning) { - next.reasoning = { ...reasoning, effort: "xhigh" }; - } - return next; + return writeEffortValue(b, "xhigh", c); } if (shouldDowngradeXHigh || shouldDowngradeMax) { @@ -146,14 +203,7 @@ export function sanitizeReasoningEffortForProvider( "REASONING_SANITIZE", `${provider}/${modelStr}: downgraded reasoning_effort ${effortStr} → high` ); - const next: Record = { ...b }; - if (hasTopLevelReasoningEffort) { - next.reasoning_effort = "high"; - } - if (reasoning) { - next.reasoning = { ...reasoning, effort: "high" }; - } - return next; + return writeEffortValue(b, "high", c); } return body; diff --git a/open-sse/executors/default.ts b/open-sse/executors/default.ts index 9fad2305a46..5ce3772f8d4 100644 --- a/open-sse/executors/default.ts +++ b/open-sse/executors/default.ts @@ -562,9 +562,7 @@ export class DefaultExecutor extends BaseExecutor { withDefaults && typeof withDefaults === "object" && !Array.isArray(withDefaults) && - (this.provider === "cerebras" || - this.provider === "mistral" || - this.provider === "nvidia") && + (this.provider === "cerebras" || this.provider === "mistral" || this.provider === "nvidia") && Object.prototype.hasOwnProperty.call(withDefaults, "client_metadata") ) { const withoutClientMetadata = { ...(withDefaults as Record) }; @@ -732,10 +730,8 @@ export class DefaultExecutor extends BaseExecutor { } } - // ClinePass reasoning models burn all of max_tokens on the thinking phase - // when the budget is too small, leaving content empty (finish_reason: - // "length"). Bump max_tokens to a safe floor when reasoning is enabled and - // the budget is undersized. CLINEPASS-GATED — no-op for every other provider. + // Reasoning models burn all of max_tokens on the thinking phase when the budget is too + // small, leaving content empty (finish_reason: "length"); applies to all providers (#6912). if (typeof withDefaults === "object" && withDefaults !== null) { this.ensureThinkingBudget(withDefaults as Record, model); } @@ -760,12 +756,10 @@ export class DefaultExecutor extends BaseExecutor { return withDefaults; } - // ClinePass / OpenRouter-style thinking models leave content empty when the - // reasoning budget consumes all of max_tokens. Bump max_tokens to a safe - // minimum only when reasoning is enabled and the budget is undersized. - // CLINEPASS-GATED: returns early for every other provider. + // Reasoning models (ClinePass, OpenRouter, etc.) leave content empty when the reasoning + // budget consumes all of max_tokens; bump max_tokens to a safe minimum when undersized. ensureThinkingBudget(body: Record, model: string): Record { - if (!body || this.provider !== "clinepass") return body; + if (!body) return body; const outboundModel = typeof body.model === "string" ? body.model : model; const entry = getRegistryEntry(this.provider); @@ -789,10 +783,15 @@ export class DefaultExecutor extends BaseExecutor { const target = Math.min(MIN_TOKENS, maxOutput); const current = body.max_tokens ?? body.max_completion_tokens; + // #6912: keep whichever token key transformRequest already set (o1/o3/o4/gpt-5 use + // max_completion_tokens) instead of re-introducing max_tokens alongside it. + const tokenKey = + body.max_completion_tokens !== undefined ? "max_completion_tokens" : "max_tokens"; + if (typeof current !== "number" || current <= 0) { - body.max_tokens = target; + body[tokenKey] = target; } else if (current < MIN_TOKENS && current < maxOutput) { - body.max_tokens = MIN_TOKENS; + body[tokenKey] = MIN_TOKENS; } return body; } diff --git a/open-sse/executors/duckduckgo-web.ts b/open-sse/executors/duckduckgo-web.ts index bf653932c78..0d5107013f9 100644 --- a/open-sse/executors/duckduckgo-web.ts +++ b/open-sse/executors/duckduckgo-web.ts @@ -8,6 +8,60 @@ import type { Session } from "../services/sessionPool/session.ts"; import { tryBackedChat } from "../services/browserBackedChat.ts"; import { sanitizeErrorMessage } from "../utils/error.ts"; +// Issue #6999: Lightweight circuit breaker for the DuckDuckGo executor. +// After CB_THRESHOLD consecutive failures (429, 5xx, or network errors), +// the breaker "opens" for CB_COOLDOWN_MS — during that window every request +// fast-fails with 503 instead of hammering the upstream. A single success +// resets the failure counter. Half-open probing happens naturally: once the +// cooldown expires the breaker closes and the next request is a real probe. +export const CB_THRESHOLD = 5; +export const CB_COOLDOWN_MS = 30_000; + +interface CircuitBreakerState { + failures: number; + openedAt: number; +} + +const circuitBreaker: CircuitBreakerState = { failures: 0, openedAt: 0 }; + +export function cbIsOpen(): boolean { + if (circuitBreaker.openedAt === 0) return false; + if (Date.now() - circuitBreaker.openedAt >= CB_COOLDOWN_MS) { + // Cooldown elapsed — half-open: allow the next request through. + circuitBreaker.openedAt = 0; + return false; + } + return true; +} + +export function cbRecordFailure(): void { + circuitBreaker.failures++; + if (circuitBreaker.failures >= CB_THRESHOLD && circuitBreaker.openedAt === 0) { + circuitBreaker.openedAt = Date.now(); + console.warn( + `[DDG-CB] Circuit breaker opened after ${circuitBreaker.failures} consecutive failures — fast-failing for ${CB_COOLDOWN_MS}ms` + ); + } +} + +export function cbRecordSuccess(): void { + if (circuitBreaker.failures > 0) { + circuitBreaker.failures = 0; + } +} + +// Test-only: direct read/write access to the module-level breaker singleton +// so tests can exercise open/half-open/closed transitions without waiting +// CB_COOLDOWN_MS in real time. Not used by production code. +export function __setDdgCircuitBreakerStateForTests(failures: number, openedAt: number): void { + circuitBreaker.failures = failures; + circuitBreaker.openedAt = openedAt; +} + +export function __getDdgCircuitBreakerStateForTests(): CircuitBreakerState { + return { ...circuitBreaker }; +} + export const DUCKDUCKGO_BASE = "https://duckduckgo.com"; // #4037: the live DuckDuckGo AI Chat backend is served from duckduckgo.com. The // status/chat fetches, Origin, and Referer must all use this host so the request's @@ -389,6 +443,13 @@ export class DuckDuckGoWebExecutor extends BaseExecutor { return errorResponse(400, "No messages provided"); } + // Issue #6999: Circuit breaker fast-fail. If DDG has been consistently + // failing, short-circuit with 503 so the combo engine can immediately + // fail over to the next provider instead of waiting for timeouts. + if (cbIsOpen()) { + return errorResponse(503, "DuckDuckGo circuit breaker open — upstream unavailable"); + } + // Browser-backed path: opt-in via OMNIROUTE_BROWSER_POOL=on or // WEB_COOKIE_USE_BROWSER=1. Routes the chat through a shared // Playwright/Cloakbrowser page so DDG's VQD challenge is solved by @@ -508,6 +569,7 @@ export class DuckDuckGoWebExecutor extends BaseExecutor { if (chatResponse.status === 429) { if (pool && session) pool.reportCooldown(session); + cbRecordFailure(); return await this.processResponse(chatResponse, isStreaming, hasTools, requestedTools); } @@ -523,6 +585,7 @@ export class DuckDuckGoWebExecutor extends BaseExecutor { if (chatResponse.status >= 500) { if (pool && session) pool.reportDead(session); + cbRecordFailure(); return errorResponse(502, "Upstream error"); } @@ -544,11 +607,13 @@ export class DuckDuckGoWebExecutor extends BaseExecutor { } } + cbRecordSuccess(); return result; } catch (error) { if (pool && session) { pool.reportCooldown(session); } + cbRecordFailure(); if (error instanceof DOMException && error.name === "AbortError") { return errorResponse(499, "Request cancelled"); diff --git a/open-sse/executors/opencode.ts b/open-sse/executors/opencode.ts index 2453199b8f1..babae8a710d 100644 --- a/open-sse/executors/opencode.ts +++ b/open-sse/executors/opencode.ts @@ -41,17 +41,37 @@ const OPENCODE_COOLDOWN_MAX_MS = 60_000; const EFFORT_LEVELS = ["low", "medium", "high", "max"] as const; /** - * Parse a DeepSeek V4 Pro model string with an effort-level suffix. + * Models on opencode-go that support effort-tier aliases. Each entry maps the + * canonical base id to the set of effort suffixes the upstream supports. + * + * - deepseek-v4-pro: all four tiers (low/medium/high/max) + * - glm-5.2: high/max only (Z.AI maps these through the reasoning plane; + * low/medium are not supported on the OpenAI transport) + * - mimo-v2.5: high/max only (same reasoning; Xiaomi MiMo does not document + * low/medium effort tiers) + */ +const EFFORT_TIERS: Record = { + "deepseek-v4-pro": EFFORT_LEVELS, + "glm-5.2": ["high", "max"], + "mimo-v2.5": ["high", "max"], +}; + +/** + * Parse a model string with an effort-level suffix. * e.g. "deepseek-v4-pro-low" → { baseModel: "deepseek-v4-pro", effort: "low" } - * Returns null if the model doesn't match the pattern. + * "glm-5.2-high" → { baseModel: "glm-5.2", effort: "high" } + * Returns null if the model doesn't match any known effort-tier pattern. */ -function parseDeepSeekEffortLevel(model: string): { baseModel: string; effort: string } | null { +export function parseEffortLevel(model: string): { baseModel: string; effort: string } | null { const m = String(model || ""); - const matchedLevel = EFFORT_LEVELS.find((level) => m.endsWith(`-${level}`)); - if (!matchedLevel) return null; - const baseModel = m.slice(0, -matchedLevel.length - 1); - if (baseModel.toLowerCase() !== "deepseek-v4-pro") return null; - return { baseModel: "deepseek-v4-pro", effort: matchedLevel }; + for (const [baseModel, levels] of Object.entries(EFFORT_TIERS)) { + for (const level of levels) { + if (m === `${baseModel}-${level}`) { + return { baseModel, effort: level }; + } + } + } + return null; } export class OpencodeExecutor extends BaseExecutor { @@ -316,7 +336,7 @@ export class OpencodeExecutor extends BaseExecutor { } if (modifiedBody && typeof modifiedBody === "object" && !Array.isArray(modifiedBody)) { const mb = modifiedBody as Record; - const parsed = parseDeepSeekEffortLevel(model); + const parsed = parseEffortLevel(model); if (parsed) { mb.model = parsed.baseModel; if (mb.reasoning_effort === undefined) { diff --git a/open-sse/executors/theoldllm.ts b/open-sse/executors/theoldllm.ts index 64e985b96c3..8acc471b3a9 100644 --- a/open-sse/executors/theoldllm.ts +++ b/open-sse/executors/theoldllm.ts @@ -108,6 +108,23 @@ export function mapModel(model: string): string { const TOKEN_SEED = "oldllm-client-2026"; const UA_PREFIX = CHROME_UA.slice(0, 20); // "Mozilla/5.0 (Windows" +type TheOldLlmProxy = { + type?: string; + host: string; + port: number; + username?: string | null; + password?: string | null; +} | null; + +interface TheOldLlmFetchDependencies { + resolveProxy: () => Promise; + runWithProxy: (proxy: TheOldLlmProxy, request: () => Promise) => Promise; + fetch: typeof fetch; + hasBlockingProxyAssignment?: () => boolean; +} + +class TheOldLlmProxyUnavailableError extends Error {} + export function generateRequestToken(): string { const n = Date.now(); const e = `${n}-${TOKEN_SEED}-${UA_PREFIX}`; @@ -127,17 +144,31 @@ export const tokenCache: { value: string; expiresAt: number } = { value: "", exp // ── Direct Node.js fetch ────────────────────────────────────────────────── -async function directFetch( +export async function fetchTheOldLlmWithProviderProxy( reqBody: Record, - signal?: AbortSignal | null + signal: AbortSignal, + dependencies?: TheOldLlmFetchDependencies ): Promise { - const controller = new AbortController(); - const timer = setTimeout(() => controller.abort(), 120_000); - const onSignal = signal ? () => controller.abort(signal.reason) : undefined; - signal?.addEventListener("abort", onSignal!, { once: true }); + let deps = dependencies; + if (!deps) { + const [ + { resolveProxyForProvider, hasBlockingProxyAssignmentForProvider }, + { runWithProxyContext }, + ] = await Promise.all([import("../../src/lib/db/proxies"), import("../utils/proxyFetch.ts")]); + deps = { + resolveProxy: () => resolveProxyForProvider("theoldllm"), + runWithProxy: runWithProxyContext, + fetch: globalThis.fetch, + hasBlockingProxyAssignment: () => hasBlockingProxyAssignmentForProvider("theoldllm"), + }; + } - try { - return await fetch(API_URL, { + const proxy = await deps.resolveProxy(); + if (!proxy && deps.hasBlockingProxyAssignment?.()) { + throw new TheOldLlmProxyUnavailableError("No active proxy is available for The Old LLM"); + } + return deps.runWithProxy(proxy, () => + deps.fetch(API_URL, { method: "POST", headers: { "Content-Type": "application/json", @@ -146,14 +177,41 @@ async function directFetch( "User-Agent": CHROME_UA, }, body: JSON.stringify(reqBody), - signal: controller.signal, - }); + signal, + }) + ); +} + +async function directFetch( + reqBody: Record, + signal?: AbortSignal | null +): Promise { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), 120_000); + const onSignal = signal ? () => controller.abort(signal.reason) : undefined; + signal?.addEventListener("abort", onSignal!, { once: true }); + + try { + // No-auth providers do not have a connection row, so chatCore cannot apply + // a connection-scoped proxy context for them. Resolve the provider/global + // assignment explicitly; otherwise The Old LLM always leaks out through the + // VPS address and Vercel's bot protection denies every model. + return await fetchTheOldLlmWithProviderProxy(reqBody, controller.signal); } finally { clearTimeout(timer); if (onSignal) signal?.removeEventListener("abort", onSignal); } } +export function isVercelMitigationResponse(response: Response, body: string): boolean { + const mitigation = response.headers.get("x-vercel-mitigated")?.toLowerCase(); + if (mitigation === "deny" || mitigation === "challenge") return true; + return ( + (response.status === 403 || response.status === 429) && + /vercel security checkpoint|"message"\s*:\s*"forbidden"/i.test(body) + ); +} + function isTokenRejected(status: number, body: string): boolean { if (status === 401 || status === 403) return true; try { @@ -211,6 +269,45 @@ function buildErrorResponse(status: number, body: string): string { }); } +function buildVercelMitigationError(): string { + return JSON.stringify({ + error: { + message: + "The Old LLM is blocked by Vercel for this server egress IP. Configure a residential provider or global proxy for 'theoldllm' and retry.", + type: "upstream_access_denied", + code: "THEOLDLLM_VERCEL_MITIGATED", + }, + }); +} + +function buildProxyUnavailableError(): string { + return JSON.stringify({ + error: { + message: + "The Old LLM proxy assignment has no active proxies. Configure or enable a proxy and retry.", + type: "proxy_unavailable", + code: "THEOLDLLM_PROXY_UNAVAILABLE", + }, + }); +} + +async function fetchUpstreamWithRetry( + reqBody: Record, + signal: AbortSignal | null | undefined, + log: ExecuteInput["log"] +): Promise<{ response: Response; body: string; vercelMitigated: boolean }> { + let response = await directFetch(reqBody, signal); + let body = await response.text(); + let vercelMitigated = isVercelMitigationResponse(response, body); + if (!vercelMitigated && isTokenRejected(response.status, body)) { + log?.warn?.("THEOLDLLM", `Token rejected (${response.status}), retrying with fresh token…`); + response = await directFetch(reqBody, signal); + body = await response.text(); + vercelMitigated = isVercelMitigationResponse(response, body); + } + return { response, body, vercelMitigated }; +} + // ── Executor ────────────────────────────────────────────────────────────── export class TheOldLlmExecutor extends BaseExecutor { @@ -237,27 +334,37 @@ export class TheOldLlmExecutor extends BaseExecutor { return body; } + private executionResult(input: ExecuteInput, response: Response, body: unknown) { + return { + response, + url: API_URL, + headers: this.buildHeaders(input.credentials), + transformedBody: body, + }; + } + async testConnection( _credentials: ProviderCredentials, _signal?: AbortSignal | null, log?: ExecuteInput["log"] ): Promise { try { - const resp = await fetch(API_URL, { - method: "POST", - headers: { - "Content-Type": "application/json", - "X-Client-Version": "3.8.4", - "X-Request-Token": generateRequestToken(), - "User-Agent": CHROME_UA, - }, - body: JSON.stringify({ + const resp = await directFetch( + { model: "GPT_5_4", messages: [{ role: "user", content: "ping" }], stream: false, - }), - signal: _signal ?? undefined, - }); + }, + _signal + ); + const body = await resp.text(); + if (!resp.ok && isVercelMitigationResponse(resp, body)) { + log?.warn?.( + "THEOLDLLM", + "Vercel blocked this egress IP; configure a residential provider proxy" + ); + return false; + } return resp.status === 200; } catch { log?.warn?.("THEOLDLLM", "testConnection network error"); @@ -297,56 +404,55 @@ export class TheOldLlmExecutor extends BaseExecutor { stream: true, }; - let upstream = await directFetch(reqBody, signal); - let finalBody = await upstream.text(); - - if (isTokenRejected(upstream.status, finalBody)) { - log?.warn?.("THEOLDLLM", `Token rejected (${upstream.status}), retrying with fresh token…`); - upstream = await directFetch(reqBody, signal); - finalBody = await upstream.text(); - } + const { + response: upstream, + body: finalBody, + vercelMitigated, + } = await fetchUpstreamWithRetry(reqBody, signal, log); if (upstream.status === 200 && finalBody) { const payload = stream ? finalBody : buildChatCompletion(parseSseContent(finalBody), model); - return { - response: new Response(encoder.encode(payload), { + return this.executionResult( + input, + new Response(encoder.encode(payload), { status: 200, headers: { "Content-Type": stream ? "text/event-stream" : "application/json", "Cache-Control": "no-cache", }, }), - url: API_URL, - headers: this.buildHeaders(input.credentials), - transformedBody: body, - }; + body + ); } - return { - response: new Response(encoder.encode(buildErrorResponse(upstream.status, finalBody)), { + const errorPayload = vercelMitigated + ? buildVercelMitigationError() + : buildErrorResponse(upstream.status, finalBody); + return this.executionResult( + input, + new Response(encoder.encode(errorPayload), { status: upstream.status, headers: { "Content-Type": "application/json" }, }), - url: API_URL, - headers: this.buildHeaders(input.credentials), - transformedBody: body, - }; + body + ); } catch (err) { + const proxyUnavailable = err instanceof TheOldLlmProxyUnavailableError; const msg = err instanceof Error ? err.message : String(err); log?.error?.("THEOLDLLM", `Executor error: ${msg}`); - return { - response: new Response( - encoder.encode( - JSON.stringify({ - error: { message: msg, type: "upstream_error", code: "EXECUTOR_ERROR" }, - }) - ), - { status: 502, headers: { "Content-Type": "application/json" } } - ), - url: API_URL, - headers: this.buildHeaders(input.credentials), - transformedBody: body, - }; + const errorPayload = proxyUnavailable + ? buildProxyUnavailableError() + : JSON.stringify({ + error: { message: msg, type: "upstream_error", code: "EXECUTOR_ERROR" }, + }); + return this.executionResult( + input, + new Response(encoder.encode(errorPayload), { + status: proxyUnavailable ? 503 : 502, + headers: { "Content-Type": "application/json" }, + }), + body + ); } } } diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 83a3eb71520..abcd3a9c0bf 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -14,6 +14,7 @@ import { buildPostCallGuardrailContext } from "./chatCore/postCallGuardrailConte import { storeSemanticCacheResponse } from "./chatCore/semanticCacheStore.ts"; import { buildNonStreamingResponseHeaders } from "./chatCore/nonStreamingResponseHeaders.ts"; import { buildNonStreamingJsonResponse } from "./chatCore/nonStreamingJsonResponse.ts"; +import { enforceOutputTokenBudget } from "./chatCore/outputTokenBudget.ts"; import { maybeConvertJsonBodyToSse } from "./chatCore/jsonBodyToSse.ts"; import { assembleStreamingResponseHeaders } from "./chatCore/streamingResponseHeaders.ts"; import { storeStreamingSemanticCacheResponse } from "./chatCore/streamingSemanticCacheStore.ts"; @@ -125,7 +126,11 @@ import { formatProviderError, sanitizeErrorMessage, } from "../utils/error.ts"; -import { reportMalformed200, detectMalformedNonStream } from "../utils/diagnostics.ts"; +import { + reportMalformed200, + detectMalformedNonStream, + describeMalformedNonStream, +} from "../utils/diagnostics.ts"; import { checkTokenLimits, recordTokenUsage, @@ -140,6 +145,7 @@ import { STREAM_READINESS_TIMEOUT_MS, ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE, STREAM_RECOVERY, + DEFAULT_MAX_TOKENS, } from "../config/constants.ts"; import { createRecoverableStream, makeContinuationBody } from "../services/streamRecovery.ts"; import { @@ -301,7 +307,12 @@ import { resolveComboContextLimit, } from "../services/contextManager.ts"; import { resolveBackgroundTaskRedirect } from "./chatCore/backgroundRedirect.ts"; -import type { CompressionConfig, CompressionPipelineStep } from "../services/compression/types.ts"; +import type { + CompressionConfig, + CompressionPipelineStep, + CompressionResult, +} from "../services/compression/types.ts"; +import { generateSessionId } from "../services/sessionManager.ts"; import { prepareWebSearchFallbackBody } from "../services/webSearchFallback.ts"; import { resolveInterceptSearch } from "@/lib/db/interceptionRules"; import { @@ -1334,7 +1345,8 @@ export async function handleChatCore({ // that selectCompressionStrategy can only partially apply via the mode string. const cacheCtx = { provider, targetFormat, model: effectiveModel, connectionCacheOverride }; const compressionConfig = resolveCacheAwareConfig(config, compressionInputBody, cacheCtx); - const result = await applyCompressionAsync(compressionInputBody, mode, { + const compressionPrincipalId = apiKeyInfo?.id ? String(apiKeyInfo.id) : undefined; + const compressionOptions = { model: effectiveModel, // #7237: feed the AUTHORITATIVE capability (model spec / models.dev sync / DB // override, with the conservative model-id fragment heuristic only as its @@ -1348,10 +1360,11 @@ export async function handleChatCore({ .supportsVision, // Rota direta oficial ('anthropic') vs agregadores: o engine omniglyph // exige 'direct' — agregadores redimensionam imagens (medido 2026-07-06). - providerTransport: provider === "anthropic" ? "direct" : "aggregator", + providerTransport: + provider === "anthropic" ? ("direct" as const) : ("aggregator" as const), config: compressionConfig, cachingContext: cacheCtx, - principalId: apiKeyInfo?.id ? String(apiKeyInfo.id) : undefined, + principalId: compressionPrincipalId, // F3.3: stream per-engine progress live (best-effort) before compression.completed. onEngineStep: (s) => { try { @@ -1375,7 +1388,52 @@ export async function handleChatCore({ // best-effort live event — never fail the request } }, - }); + }; + const runCompression = (input: Record) => + applyCompressionAsync(input, mode, compressionOptions); + let result: CompressionResult; + if (compressionConfig.liveZone?.enabled === true) { + const { applyLiveZoneCompression } = await import("../services/compression/liveZone.ts"); + const explicitSessionId = + clientRawRequest?.headers && typeof clientRawRequest.headers.get === "function" + ? clientRawRequest.headers.get("x-omniroute-session-id") + : getHeaderValueCaseInsensitive( + clientRawRequest?.headers ?? null, + "x-omniroute-session-id" + ); + const liveZoneSessionId = + explicitSessionId || + generateSessionId(compressionInputBody, { + provider, + connectionId: getCurrentConnectionId() ?? undefined, + }) || + undefined; + result = await applyLiveZoneCompression( + compressionInputBody, + { + principalId: compressionPrincipalId, + sessionId: liveZoneSessionId, + variant: { + mode, + provider, + model: effectiveModel, + config: compressionConfig, + cachePrefix: { + system: compressionInputBody.system, + systemInstruction: compressionInputBody.systemInstruction, + system_instruction: compressionInputBody.system_instruction, + instructions: compressionInputBody.instructions, + tools: compressionInputBody.tools, + toolChoice: compressionInputBody.tool_choice, + }, + }, + ttlMinutes: compressionConfig.cacheMinutes, + }, + runCompression + ); + } else { + result = await runCompression(compressionInputBody); + } if (result.stats) { const annotation = formatCompressionAnnotation(result.stats); if (annotation) { @@ -1441,6 +1499,7 @@ export async function handleChatCore({ cavemanOutputModeIntensity, log, }); + await compressionAnalyticsWritePromise; } else { // Compression was attempted (mode active, engines ran) but produced no // recordable saving — e.g. a Stacked RTK→Caveman pipeline on already-compact @@ -1463,6 +1522,7 @@ export async function handleChatCore({ }, "no_savings" ); + await compressionAnalyticsWritePromise; } if (result.compressed) { @@ -1493,6 +1553,7 @@ export async function handleChatCore({ cavemanOutputModeIntensity, log, }); + await compressionAnalyticsWritePromise; } emitOutputStyleTelemetry({ outputStyleResult, @@ -1632,6 +1693,56 @@ export async function handleChatCore({ ); } + // Re-check the concrete target after all compression passes. Combo compatibility + // filtering is advisory and may preserve an all-incompatible pool; this is the + // hard boundary that prevents a too-large prompt (or a negative token budget) + // from reaching an OpenAI-compatible upstream such as NVIDIA NIM. + const finalCompressionBody = body + ? adaptBodyForCompression(body as Record).body + : null; + const finalMessages = + finalCompressionBody?.messages || + body?.contents || + body?.request?.contents || + (body?.input && typeof body.input === "object" && !Array.isArray(body.input) + ? body.input + : []); + const finalEstimatedInputTokens = + estimateTokens(finalMessages) + + (Array.isArray(body?.tools) ? estimateTokens(body.tools) : 0) + + estimateTokens(body?.system) + + estimateTokens(body?.instructions); + const finalContextLimit = getTokenLimit(provider, effectiveModel); + const outputBudget = enforceOutputTokenBudget( + body as Record, + finalEstimatedInputTokens, + finalContextLimit, + targetFormat === FORMATS.CLAUDE && sourceFormat !== FORMATS.CLAUDE ? DEFAULT_MAX_TOKENS : 0 + ); + if (!outputBudget.ok) { + const message = + `Input exceeds the context window for ${provider}/${effectiveModel}: ` + + `estimated ${outputBudget.estimatedInputTokens} input tokens, limit ${outputBudget.contextLimit}. ` + + "Reduce the prompt or route to a model with a larger context window."; + log?.warn?.("CONTEXT", message); + trackPendingRequest(model, provider, connectionId, false); + return createErrorResult( + HTTP_STATUS.BAD_REQUEST, + message, + null, + "context_length_exceeded", + "invalid_request_error" + ); + } + if (outputBudget.adjustedFields.length > 0) { + log?.info?.( + "CONTEXT", + `Adjusted invalid or oversized output token fields (${outputBudget.adjustedFields.join(", ")}); ` + + `${outputBudget.availableOutputTokens} tokens remain for output` + ); + } + body = outputBudget.body; + let translatedBody = body; const isClaudePassthrough = sourceFormat === FORMATS.CLAUDE && targetFormat === FORMATS.CLAUDE; const isClaudeCodeCompatible = isClaudeCodeCompatibleProvider(provider); @@ -3384,7 +3495,13 @@ export async function handleChatCore({ // otherwise degenerate into a 429 rate-limit storm). Connection stays // active since only the specific model is unavailable. (#6827) const notFoundCooldownMs = COOLDOWN_MS.notFound; - lockModel(provider, errorConnectionId, currentModel, "model_not_found", notFoundCooldownMs); + lockModel( + provider, + errorConnectionId, + currentModel, + "model_not_found", + notFoundCooldownMs + ); console.warn( `[provider] Node ${errorConnectionId} model not found (${statusCode}) for ${currentModel} - locking model for ${Math.ceil(notFoundCooldownMs / 1000)}s (connection stays active)` ); @@ -4070,7 +4187,11 @@ export async function handleChatCore({ connectionId, status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}`, }).catch(() => {}); - const malformedMessage = `[${provider}/${model}] returned an empty response (no usable choices/output)`; + const malformed = describeMalformedNonStream(translatedResponse, malformedTranslatedReason); + const malformedMessage = `[${provider}/${model}] ${malformed.message}`; + const malformedClientBody = buildErrorBody(HTTP_STATUS.BAD_GATEWAY, malformedMessage); + malformedClientBody.error.code = malformed.code; + malformedClientBody.error.type = malformed.type; persistAttemptLogs({ status: HTTP_STATUS.BAD_GATEWAY, tokens: usage, @@ -4079,14 +4200,20 @@ export async function handleChatCore({ providerResponse: looksLikeSSE ? { _streamed: true, _format: "sse-json", summary: responseBody } : responseBody, - clientResponse: buildErrorBody(HTTP_STATUS.BAD_GATEWAY, malformedMessage), + clientResponse: malformedClientBody, claudeCacheMeta: claudePromptCacheLogMeta, claudeCacheUsageMeta: cacheUsageLogMeta, cacheSource: "upstream", }); persistFailureUsage(HTTP_STATUS.BAD_GATEWAY, "malformed_translated_response"); trackPendingRequest(model, provider, pendingConnId, false); - return createErrorResult(HTTP_STATUS.BAD_GATEWAY, malformedMessage); + return createErrorResult( + HTTP_STATUS.BAD_GATEWAY, + malformedMessage, + null, + malformed.code, + malformed.type + ); } // ── Phase 9.1: Cache store (non-streaming, temp=0) ── diff --git a/open-sse/handlers/chatCore/cavemanOutputAnalytics.ts b/open-sse/handlers/chatCore/cavemanOutputAnalytics.ts index fa3a778d9cd..18307628ce6 100644 --- a/open-sse/handlers/chatCore/cavemanOutputAnalytics.ts +++ b/open-sse/handlers/chatCore/cavemanOutputAnalytics.ts @@ -5,11 +5,11 @@ * Extracted from handleChatCore's request-setup compression path: when only the caveman output * mode was applied (no upstream compression run recorded a row), persist a single analytics row so * output-caveman runs still surface in compression analytics. Best-effort — returns the write - * promise (the caller assigns it to compressionAnalyticsWritePromise) and swallows its own errors. - * Behaviour is byte-identical to the previous inline block. + * promise so the caller can finish persistence before dispatch; errors remain non-fatal but are + * logged at warning level. */ -type LoggerLike = { debug?: (...args: unknown[]) => void } | null | undefined; +type LoggerLike = { warn?: (...args: unknown[]) => void } | null | undefined; export function writeCavemanOutputAnalytics(args: { comboName: string | null | undefined; @@ -37,7 +37,7 @@ export function writeCavemanOutputAnalytics(args: { output_mode: args.cavemanOutputModeIntensity, }); } catch (err) { - args.log?.debug?.( + args.log?.warn?.( "COMPRESSION", "Caveman output analytics write skipped: " + (err instanceof Error ? err.message : String(err)) diff --git a/open-sse/handlers/chatCore/compressionAnalyticsWrite.ts b/open-sse/handlers/chatCore/compressionAnalyticsWrite.ts index 21b21b33c02..ec68aff23cc 100644 --- a/open-sse/handlers/chatCore/compressionAnalyticsWrite.ts +++ b/open-sse/handlers/chatCore/compressionAnalyticsWrite.ts @@ -4,15 +4,21 @@ * * Extracted from handleChatCore's request-setup compression path: persist the per-run compression * analytics row (cost saved, RTK raw-output pointers) plus the per-engine breakdown of a stacked - * run. Returns the write promise (the caller assigns it to compressionAnalyticsWritePromise) and - * swallows its own errors — best-effort, off the hot path, never throws into a request. Behaviour - * is byte-identical to the previous inline block. Split into small builders so each stays under the - * complexity cap. + * run. Returns the write promise so the caller can finish persistence before dispatching the + * upstream request. Errors remain best-effort and never throw into a request, but they are logged + * at warning level so a broken analytics path is observable. Split into small builders so each + * stays under the complexity cap. */ import { type CompressionStats } from "../../services/compression/stats.ts"; -type LoggerLike = { debug?: (...args: unknown[]) => void } | null | undefined; +type LoggerLike = + | { + debug?: (...args: unknown[]) => void; + warn?: (...args: unknown[]) => void; + } + | null + | undefined; type WriteOpts = { stats: CompressionStats; @@ -30,6 +36,12 @@ type WriteOpts = { type RtkPointer = { id?: string | null; bytes?: number | null }; +type CalculateCost = typeof import("@/lib/usage/costCalculator").calculateCost; + +type WriteDependencies = { + calculateCost?: CalculateCost; +}; + function buildRtkPointerFields(rtkPointers: RtkPointer[]) { return { rtk_raw_output_pointer: rtkPointers[0]?.id ?? null, @@ -109,7 +121,7 @@ export function writeCompressionSkip(opts: WriteOpts, skipReason: string): Promi skip_reason: skipReason, }); } catch (err) { - opts.log?.debug?.( + opts.log?.warn?.( "COMPRESSION", "Compression skip-analytics write skipped: " + (err instanceof Error ? err.message : String(err)) @@ -118,29 +130,42 @@ export function writeCompressionSkip(opts: WriteOpts, skipReason: string): Promi })(); } -export function writeCompressionAnalytics(opts: WriteOpts): Promise { +export function writeCompressionAnalytics( + opts: WriteOpts, + dependencies: WriteDependencies = {} +): Promise { return (async () => { try { - const { insertCompressionAnalyticsRow, insertCompressionEngineBreakdown } = await import( - "@/lib/db/compressionAnalytics" - ); - const { calculateCost } = await import("@/lib/usage/costCalculator"); + const { insertCompressionAnalyticsRow, insertCompressionEngineBreakdown } = + await import("@/lib/db/compressionAnalytics"); const { stats } = opts; const tokensSaved = Math.max(0, stats.originalTokens - stats.compressedTokens); const rtkPointers = (stats.rtkRawOutputPointers ?? []) as RtkPointer[]; - const estimatedUsdSaved = await calculateCost( - opts.provider ?? "", - opts.effectiveModel ?? "", - { input: tokensSaved }, - { serviceTier: opts.effectiveServiceTier } + let estimatedUsdSaved = 0; + try { + const calculateCost = + dependencies.calculateCost ?? (await import("@/lib/usage/costCalculator")).calculateCost; + estimatedUsdSaved = await calculateCost( + opts.provider ?? "", + opts.effectiveModel ?? "", + { input: tokensSaved }, + { serviceTier: opts.effectiveServiceTier } + ); + } catch (err) { + opts.log?.debug?.( + "COMPRESSION", + "Compression cost estimate skipped: " + (err instanceof Error ? err.message : String(err)) + ); + } + insertCompressionAnalyticsRow( + buildAnalyticsRow(opts, tokensSaved, rtkPointers, estimatedUsdSaved) ); - insertCompressionAnalyticsRow(buildAnalyticsRow(opts, tokensSaved, rtkPointers, estimatedUsdSaved)); const breakdownRows = buildEngineBreakdownRows(stats, opts.skillRequestId); if (breakdownRows.length > 0) { insertCompressionEngineBreakdown(breakdownRows); } } catch (err) { - opts.log?.debug?.( + opts.log?.warn?.( "COMPRESSION", "Compression analytics write skipped: " + (err instanceof Error ? err.message : String(err)) ); diff --git a/open-sse/handlers/chatCore/outputTokenBudget.ts b/open-sse/handlers/chatCore/outputTokenBudget.ts new file mode 100644 index 00000000000..2d26d488c2a --- /dev/null +++ b/open-sse/handlers/chatCore/outputTokenBudget.ts @@ -0,0 +1,117 @@ +export const OUTPUT_TOKEN_FIELDS = [ + "max_tokens", + "max_completion_tokens", + "max_output_tokens", +] as const; + +export type OutputTokenBudgetResult = + | { + ok: true; + body: Record; + availableOutputTokens: number; + adjustedFields: string[]; + } + | { + ok: false; + estimatedInputTokens: number; + contextLimit: number; + }; + +type OutputTokenAdjustment = { field: string; value?: number; remove?: boolean }; + +function getOutputTokenAdjustment( + field: string, + value: unknown, + availableOutputTokens: number +): OutputTokenAdjustment | null { + if (typeof value !== "number") return null; + if (!Number.isFinite(value) || value <= 0) return { field, remove: true }; + + const capped = Math.min(Math.floor(value), availableOutputTokens); + return capped === value ? null : { field, value: capped }; +} + +function hasTranslatorOutputTokenLimit(body: Record): boolean { + return ["max_tokens", "max_completion_tokens"].some((field) => { + const value = body[field]; + return typeof value === "number" && Number.isFinite(value) && value > 0; + }); +} + +function adjustOutputTokenFields( + body: Record, + availableOutputTokens: number +): Pick, "body" | "adjustedFields"> { + const adjustments = OUTPUT_TOKEN_FIELDS.map((field) => + getOutputTokenAdjustment(field, body[field], availableOutputTokens) + ).filter((adjustment): adjustment is OutputTokenAdjustment => adjustment !== null); + if (adjustments.length === 0) return { body, adjustedFields: [] }; + + const nextBody = { ...body }; + for (const adjustment of adjustments) { + if (adjustment.remove) delete nextBody[adjustment.field]; + else nextBody[adjustment.field] = adjustment.value; + } + + return { body: nextBody, adjustedFields: adjustments.map(({ field }) => field) }; +} + +/** + * Enforce the target model's context budget immediately before translation. + * + * Compression and combo selection are best-effort: a request may still be too + * large for a concrete target, and some OpenAI-compatible gateways derive an + * internal max_tokens value by subtracting the prompt from the context window. + * Reject that target locally instead of allowing the derived value to become + * negative upstream. Positive client limits are capped to the remaining room; + * invalid numeric limits are removed. + */ +export function enforceOutputTokenBudget( + body: Record | null | undefined, + estimatedInputTokens: number, + contextLimit: number, + defaultOutputTokens = 0 +): OutputTokenBudgetResult { + const normalizedInputTokens = Math.max(0, Math.ceil(estimatedInputTokens)); + const normalizedContextLimit = Math.max(1, Math.floor(contextLimit)); + const normalizedDefaultOutputTokens = Math.max(0, Math.floor(defaultOutputTokens)); + const availableOutputTokens = normalizedContextLimit - normalizedInputTokens; + + if (availableOutputTokens < 1) { + return { + ok: false, + estimatedInputTokens: normalizedInputTokens, + contextLimit: normalizedContextLimit, + }; + } + + if (!body) { + if (normalizedDefaultOutputTokens > availableOutputTokens) { + return { + ok: false, + estimatedInputTokens: normalizedInputTokens, + contextLimit: normalizedContextLimit, + }; + } + return { + ok: true, + body: {}, + availableOutputTokens, + adjustedFields: [], + }; + } + + if ( + normalizedDefaultOutputTokens > availableOutputTokens && + !hasTranslatorOutputTokenLimit(body) + ) { + return { + ok: false, + estimatedInputTokens: normalizedInputTokens, + contextLimit: normalizedContextLimit, + }; + } + + const adjusted = adjustOutputTokenFields(body, availableOutputTokens); + return { ok: true, ...adjusted, availableOutputTokens }; +} diff --git a/open-sse/handlers/chatCore/streamingPipeline.ts b/open-sse/handlers/chatCore/streamingPipeline.ts index d67803918b1..2a6a7c00bbb 100644 --- a/open-sse/handlers/chatCore/streamingPipeline.ts +++ b/open-sse/handlers/chatCore/streamingPipeline.ts @@ -23,6 +23,17 @@ import { createPiiSseTransform as defaultPiiSse } from "@/lib/streamingPiiTransf import { isFeatureFlagEnabled as defaultFeatureFlag } from "@/shared/utils/featureFlags"; import { OMNIROUTE_RESPONSE_HEADERS } from "@/shared/constants/headers"; import { SSE_HEARTBEAT_INTERVAL_MS } from "../../config/constants.ts"; +/** + * Pipeline assembly instrumentation — performance.mark() along the SSE hot path. + * Marks are visible to Node.js perf_hooks consumers and DevTools' Performance + * panel when NODE_OPTIONS=--enable-node-performance-clinician or similar. + * + * Each call to assembleStreamingPipeline creates one measure record: + * "omni-pipeline" — wall-clock duration of the full transform chain assembly. + */ +const PIPELINE_START = "omni-pipeline-start"; +const PIPELINE_END = "omni-pipeline-end"; +const PIPELINE_MEASURE = "omni-pipeline"; type HeadersLike = Headers | Record | null | undefined; @@ -61,6 +72,10 @@ export function assembleStreamingPipeline( }, deps: StreamingPipelineDeps = DEFAULT_DEPS ) { + performance.clearMarks(PIPELINE_START); + performance.clearMarks(PIPELINE_END); + performance.clearMeasures(PIPELINE_MEASURE); + performance.mark(PIPELINE_START); // ── Phase 9.3: Progress tracking (opt-in) ── const progressEnabled = deps.wantsProgress(args.clientRawRequestHeaders); let finalStream; @@ -77,7 +92,9 @@ export function assembleStreamingPipeline( } if (progressEnabled) { - const progressTransform = deps.createProgressTransform({ signal: args.streamController.signal }); + const progressTransform = deps.createProgressTransform({ + signal: args.streamController.signal, + }); // Chain: provider → transform → progress → client finalStream = piiStream.pipeThrough(progressTransform); args.responseHeaders[OMNIROUTE_RESPONSE_HEADERS.progress] = "enabled"; @@ -95,5 +112,7 @@ export function assembleStreamingPipeline( if (args.echoModel) { finalStream = finalStream.pipeThrough(deps.createModelEchoTransform(args.echoModel)); } + performance.mark(PIPELINE_END); + performance.measure(PIPELINE_MEASURE, PIPELINE_START, PIPELINE_END); return finalStream; } diff --git a/open-sse/handlers/imageGeneration.ts b/open-sse/handlers/imageGeneration.ts index 9e0adbca126..c0d3de0ec1f 100644 --- a/open-sse/handlers/imageGeneration.ts +++ b/open-sse/handlers/imageGeneration.ts @@ -54,6 +54,7 @@ import { handleHyperbolicImageGeneration } from "./imageGeneration/providers/hyp import { handleHuggingFaceImageGeneration } from "./imageGeneration/providers/huggingface.ts"; import { handleComfyUIImageGeneration } from "./imageGeneration/providers/comfyUI.ts"; import { handleImagen3ImageGeneration } from "./imageGeneration/providers/imagen3.ts"; +import { handleGoogleImagenGeneration } from "./imageGeneration/providers/googleImagen.ts"; import { handleIdeogramImageGeneration } from "./imageGeneration/providers/ideogram.ts"; import { handleHaiperImageGeneration } from "./imageGeneration/providers/haiper.ts"; import { handleLeonardoImageGeneration } from "./imageGeneration/providers/leonardo.ts"; @@ -375,6 +376,17 @@ export async function handleImageGeneration({ }); } + if (providerConfig.format === "google-imagen") { + return handleGoogleImagenGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + if (providerConfig.format === "hyperbolic") { return handleHyperbolicImageGeneration({ model, diff --git a/open-sse/handlers/imageGeneration/providers/googleImagen.ts b/open-sse/handlers/imageGeneration/providers/googleImagen.ts new file mode 100644 index 00000000000..b4b2ea67d86 --- /dev/null +++ b/open-sse/handlers/imageGeneration/providers/googleImagen.ts @@ -0,0 +1,147 @@ +// Google AI Studio (Gemini API) Imagen image generation. +// +// Unlike the antigravity "gemini-image" format (which wraps generateContent in a +// Cloud Code envelope), the Imagen family on generativelanguage.googleapis.com uses +// the dedicated ":predict" endpoint with an instances/parameters body and returns +// base64 image bytes under `predictions[].bytesBase64Encoded`. +// +// Docs: https://ai.google.dev/gemini-api/docs/imagen (Imagen requires a billing- +// enabled Google project; free-tier keys get 403 / quota 0.) + +import { saveCallLog } from "@/lib/usageDb"; +import { mapImageSize } from "../../../translator/image/sizeMapper.ts"; +import { sanitizeErrorMessage } from "../../../utils/error.ts"; + +// Only the Imagen family routes through :predict. Other gemini image models +// (gemini-*-flash-image / nano-banana) use generateContent and belong on the chat +// route, so they must not be dispatched here. +export function isImagenModel(model) { + return /^imagen-/i.test(String(model || "")); +} + +/** + * Build the Imagen :predict request body from an OpenAI-style image request. + * Pure — no I/O — so it can be unit-tested without live credentials. + */ +export function buildImagenPredictBody(body) { + const prompt = typeof body?.prompt === "string" ? body.prompt : String(body?.prompt ?? ""); + const n = Number(body?.n); + const sampleCount = Number.isFinite(n) && n > 0 ? Math.min(Math.floor(n), 4) : 1; + return { + instances: [{ prompt }], + parameters: { + sampleCount, + aspectRatio: mapImageSize(body?.aspect_ratio || body?.size), + }, + }; +} + +/** + * Normalize an Imagen :predict response into the OpenAI image-generation shape + * ({ created, data: [{ b64_json, revised_prompt }] }). Pure — unit-testable. + */ +export function parseImagenPredictResponse(data, prompt) { + const predictions = Array.isArray(data?.predictions) ? data.predictions : []; + const images = []; + for (const p of predictions) { + const b64 = p?.bytesBase64Encoded ?? p?.b64_json ?? p?.image ?? null; + if (typeof b64 === "string" && b64.length > 0) { + images.push({ b64_json: b64, revised_prompt: prompt }); + } + } + return { created: Math.floor(Date.now() / 1000), data: images }; +} + +export async function handleGoogleImagenGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials?.apiKey || credentials?.accessToken || ""; + const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); + + if (!isImagenModel(model)) { + return { + success: false, + status: 400, + error: `Model ${model} is not an Imagen model. Gemini flash-image models route through /v1/chat/completions, not /v1/images/generations.`, + }; + } + + const upstreamBody = buildImagenPredictBody(body); + // baseUrl is https://generativelanguage.googleapis.com/v1beta/models + const url = `${providerConfig.baseUrl.replace(/\/$/, "")}/${model}:predict`; + + if (log) { + log.info( + "IMAGE", + `${provider}/${model} (google-imagen) | prompt: "${prompt.slice(0, 60)}..." | aspectRatio: ${upstreamBody.parameters.aspectRatio}` + ); + } + + try { + const response = await fetch(url, { + method: "POST", + headers: { + "Content-Type": "application/json", + // Key travels in the header, never the URL, so it stays out of logs. + "x-goog-api-key": token, + }, + body: JSON.stringify(upstreamBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + const safeError = sanitizeErrorMessage(errorText); + if (log) log.error("IMAGE", `${provider} error ${response.status}: ${safeError.slice(0, 200)}`); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: response.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: safeError.slice(0, 500), + }).catch(() => {}); + + return { success: false, status: response.status, error: safeError }; + } + + const data = await response.json(); + const normalized = parseImagenPredictResponse(data, prompt); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: normalized.data.length }, + }).catch(() => {}); + + return { success: true, data: normalized }; + } catch (err) { + const errMsg = err instanceof Error ? err.message : String(err); + if (log) log.error("IMAGE", `${provider} fetch error: ${errMsg}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errMsg, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage(errMsg)}`, + }; + } +} diff --git a/open-sse/mcp-server/README.md b/open-sse/mcp-server/README.md index df32ebc7f28..1ef5f8aa5f5 100644 --- a/open-sse/mcp-server/README.md +++ b/open-sse/mcp-server/README.md @@ -1,8 +1,8 @@ # OmniRoute MCP Server -> **Model Context Protocol server** that exposes OmniRoute's gateway intelligence as **37 tools** for AI agents. +> **Model Context Protocol server** that exposes OmniRoute's gateway intelligence as **104 tools** for AI agents. > -> **Source of truth for the full tool catalog and REST surface:** [`docs/frameworks/MCP-SERVER.md`](../../docs/MCP-SERVER.md). This README focuses on architecture, configuration, and integration examples; the catalog below is a summary subset. +> **Source of truth for the full tool catalog and REST surface:** [`docs/frameworks/MCP-SERVER.md`](../../docs/frameworks/MCP-SERVER.md). This README focuses on architecture, configuration, and integration examples; the catalog below is a summary subset. The MCP Server allows any AI agent (Claude Desktop, Cursor, VS Code Copilot, custom agents) to **monitor, control, and optimize** the OmniRoute AI gateway programmatically. @@ -20,7 +20,7 @@ The MCP Server allows any AI agent (Claude Desktop, Cursor, VS Code Copilot, cus ┌──────────────────────────────────────────────────────────────────┐ │ OmniRoute MCP Server │ │ ┌──────────────┐ ┌─────────────────┐ ┌────────────────────┐ │ -│ │ Scope │ │ 37 MCP Tools │ │ Audit Logger │ │ +│ │ Scope │ │ 104 MCP Tools │ │ Audit Logger │ │ │ │ Enforcement │──│ (core + memory │──│ (SHA-256/SQLite) │ │ │ │ │ │ + skills + …) │ │ │ │ │ └──────────────┘ └────────┬────────┘ └────────────────────┘ │ @@ -157,6 +157,16 @@ omniroute --mcp | 25 | `omniroute_set_compression_engine` | `write:compression` | Set Caveman, RTK, or stacked compression mode and pipeline | | 26 | `omniroute_list_compression_combos` | `read:compression` | List named compression combos and routing assignments | | 27 | `omniroute_compression_combo_stats` | `read:compression` | Read analytics grouped by compression combo and engine | +| 28 | `omniroute_ccr_store` | `write:compression` | Store content in the caller-isolated in-memory CCR store | +| 29 | `omniroute_ccr_retrieve` | `read:compression` | Retrieve full or ranged caller-owned CCR content | +| 30 | `omniroute_ccr_inspect` | `read:compression` | Inspect CCR metadata without returning content | +| 31 | `omniroute_ccr_list` | `read:compression` | List paginated caller-owned CCR metadata | +| 32 | `omniroute_ccr_delete` | `write:compression` | Delete a caller-owned CCR block | +| 33 | `omniroute_ccr_stats` | `read:compression` | Report caller usage, bounded-store limits, and lifecycle counters | + +CCR storage is bounded and in-memory only: 2 MiB per block, 16 MiB per principal, 64 MiB global, +with a 24-hour default TTL. Full MCP retrieval is capped at 256 KiB; larger blocks use ranged or +grep retrieval. All lifecycle operations are isolated by the authenticated caller principal. MCP listable metadata descriptions are compressed at registration/list time when description compression is enabled. `omniroute_compression_status` exposes those savings separately as diff --git a/open-sse/mcp-server/__tests__/toolSearch.catalog.test.ts b/open-sse/mcp-server/__tests__/toolSearch.catalog.test.ts index e0bfa82f345..b5a982e548a 100644 --- a/open-sse/mcp-server/__tests__/toolSearch.catalog.test.ts +++ b/open-sse/mcp-server/__tests__/toolSearch.catalog.test.ts @@ -18,10 +18,9 @@ describe("getAllToolDefinitions", () => { const names = all.map((t) => t.name); expect(new Set(names).size).toBe(names.length); }); - it("includes compressionTools-only entries (omniroute_ccr_retrieve, not in MCP_TOOLS)", () => { - // Regression: compressionTools carries omniroute_ccr_retrieve, which is absent from - // MCP_TOOLS — if the collection is dropped from the catalog, tool_search can never - // surface it. Guards against the catalog omission caught in core review. - expect(all.find((t) => t.name === "omniroute_ccr_retrieve")).toBeTruthy(); + it("includes every canonical CCR lifecycle tool", () => { + for (const name of ["store", "retrieve", "inspect", "list", "delete", "stats"]) { + expect(all.find((tool) => tool.name === `omniroute_ccr_${name}`)).toBeTruthy(); + } }); }); diff --git a/open-sse/mcp-server/schemas/ccrTools.ts b/open-sse/mcp-server/schemas/ccrTools.ts new file mode 100644 index 00000000000..18630c0b3ca --- /dev/null +++ b/open-sse/mcp-server/schemas/ccrTools.ts @@ -0,0 +1,202 @@ +import { z } from "zod"; + +import type { McpToolDefinition } from "./toolDefinition.ts"; + +const ccrHash = z + .string() + .regex(/^[a-f0-9]{24}$/i) + .describe("24-hex content hash from a CCR marker or ccr:// URI"); + +export const ccrEntryMetadataOutput = z.object({ + hash: ccrHash, + bytes: z.number().int().nonnegative(), + chars: z.number().int().nonnegative(), + lines: z.number().int().nonnegative(), + contentType: z.string(), + source: z.enum(["compression", "mcp", "ionizer", "session-dedup"]), + createdAt: z.number().int().nonnegative(), + lastAccessedAt: z.number().int().nonnegative(), + expiresAt: z.number().int().nonnegative(), + retrievalCount: z.number().int().nonnegative(), +}); + +export const ccrReferenceOutput = z.object({ + hash: ccrHash, + uri: z.string().startsWith("ccr://"), + marker: z.string(), +}); + +export const ccrStoreInput = z.object({ + content: z + .string() + .min(1) + .refine((content) => Buffer.byteLength(content, "utf8") <= 2 * 1024 * 1024, { + message: "Content exceeds the 2 MiB UTF-8 CCR block limit", + }) + .describe("Verbatim content to keep in the in-memory CCR store (maximum 2 MiB UTF-8)"), + contentType: z.string().trim().min(1).max(128).optional(), + ttlSeconds: z + .number() + .int() + .min(60) + .max(7 * 24 * 60 * 60) + .optional(), +}); + +export const ccrStoreOutput = z.union([ + z.object({ + stored: z.literal(true), + reference: ccrReferenceOutput, + metadata: ccrEntryMetadataOutput, + }), + z.object({ + stored: z.literal(false), + reason: z.enum(["block_too_large", "principal_budget_exceeded", "global_budget_exceeded"]), + }), +]); + +export const ccrStoreTool: McpToolDefinition = { + name: "omniroute_ccr_store", + description: + "Store verbatim content in the caller-isolated in-memory CCR store and return a ccr:// reference plus the compatible CCR marker. Entries expire automatically and are not persisted across restarts.", + inputSchema: ccrStoreInput, + outputSchema: ccrStoreOutput, + scopes: ["write:compression"], + auditLevel: "basic", + phase: 2, + sourceEndpoints: [], +}; + +export const ccrRetrieveInput = z.object({ + hash: ccrHash, + mode: z.enum(["full", "head", "tail", "lines", "grep", "stats"]).optional(), + n: z.number().int().positive().max(10_000).optional(), + start: z.number().int().positive().optional(), + end: z.number().int().positive().optional(), + pattern: z.string().max(512).optional(), + unique: z.boolean().optional(), +}); + +export const ccrRetrieveOutput = z.union([ + z.object({ + found: z.literal(false), + error: z.string(), + }), + z.object({ + found: z.literal(true), + metadata: ccrEntryMetadataOutput, + content: z.string().optional(), + tooLargeForFull: z.boolean().optional(), + suggestedModes: z.array(z.enum(["head", "tail", "lines", "grep", "stats"])).optional(), + error: z.string().optional(), + }), +]); + +export const ccrRetrieveTool: McpToolDefinition = + { + name: "omniroute_ccr_retrieve", + description: + "Retrieve caller-owned CCR content by hash. Full MCP responses are capped at 256 KiB; use head, tail, lines, grep, or stats for larger blocks.", + inputSchema: ccrRetrieveInput, + outputSchema: ccrRetrieveOutput, + scopes: ["read:compression"], + auditLevel: "basic", + phase: 2, + sourceEndpoints: ["/api/compression/retrieve"], + }; + +export const ccrInspectInput = z.object({ hash: ccrHash }); +export const ccrInspectOutput = z.union([ + z.object({ found: z.literal(false) }), + z.object({ + found: z.literal(true), + reference: ccrReferenceOutput, + metadata: ccrEntryMetadataOutput, + }), +]); +export const ccrInspectTool: McpToolDefinition = { + name: "omniroute_ccr_inspect", + description: "Inspect metadata for a caller-owned CCR block without returning its content.", + inputSchema: ccrInspectInput, + outputSchema: ccrInspectOutput, + scopes: ["read:compression"], + auditLevel: "basic", + phase: 2, + sourceEndpoints: [], +}; + +export const ccrListInput = z.object({ + offset: z.number().int().nonnegative().optional(), + limit: z.number().int().min(1).max(100).optional(), +}); +export const ccrListOutput = z.object({ + entries: z.array(z.object({ reference: ccrReferenceOutput, metadata: ccrEntryMetadataOutput })), + total: z.number().int().nonnegative(), + offset: z.number().int().nonnegative(), + limit: z.number().int().positive(), + hasMore: z.boolean(), +}); +export const ccrListTool: McpToolDefinition = { + name: "omniroute_ccr_list", + description: "List paginated metadata for CCR blocks owned by the current caller.", + inputSchema: ccrListInput, + outputSchema: ccrListOutput, + scopes: ["read:compression"], + auditLevel: "basic", + phase: 2, + sourceEndpoints: [], +}; + +export const ccrDeleteInput = z.object({ hash: ccrHash }); +export const ccrDeleteOutput = z.object({ deleted: z.boolean() }); +export const ccrDeleteTool: McpToolDefinition = { + name: "omniroute_ccr_delete", + description: "Delete a caller-owned block from the in-memory CCR store.", + inputSchema: ccrDeleteInput, + outputSchema: ccrDeleteOutput, + scopes: ["write:compression"], + auditLevel: "basic", + phase: 2, + sourceEndpoints: [], +}; + +export const ccrStatsInput = z.object({}); +export const ccrStatsOutput = z.object({ + storage: z.literal("memory"), + entries: z.number().int().nonnegative(), + bytes: z.number().int().nonnegative(), + limits: z.object({ + maxEntries: z.number().int().positive(), + maxBlockBytes: z.number().int().positive(), + maxPrincipalBytes: z.number().int().positive(), + maxGlobalBytes: z.number().int().positive(), + defaultTtlSeconds: z.number().int().positive(), + maxTtlSeconds: z.number().int().positive(), + maxMcpFullBytes: z.number().int().positive(), + }), + lifecycle: z.object({ + expiredEvictions: z.number().int().nonnegative(), + capacityEvictions: z.number().int().nonnegative(), + rejectedStores: z.number().int().nonnegative(), + }), +}); +export const ccrStatsTool: McpToolDefinition = { + name: "omniroute_ccr_stats", + description: + "Return caller-scoped CCR entry and byte usage, lifecycle counters, and in-memory store limits.", + inputSchema: ccrStatsInput, + outputSchema: ccrStatsOutput, + scopes: ["read:compression"], + auditLevel: "basic", + phase: 2, + sourceEndpoints: [], +}; + +export const CCR_MCP_TOOLS = [ + ccrStoreTool, + ccrRetrieveTool, + ccrInspectTool, + ccrListTool, + ccrDeleteTool, + ccrStatsTool, +] as const; diff --git a/open-sse/mcp-server/schemas/index.ts b/open-sse/mcp-server/schemas/index.ts index e233dd03baf..fe9df69ff2d 100644 --- a/open-sse/mcp-server/schemas/index.ts +++ b/open-sse/mcp-server/schemas/index.ts @@ -69,6 +69,26 @@ export { cacheFlushInput, cacheFlushOutput, cacheFlushTool, + ccrEntryMetadataOutput, + ccrReferenceOutput, + ccrStoreInput, + ccrStoreOutput, + ccrStoreTool, + ccrRetrieveInput, + ccrRetrieveOutput, + ccrRetrieveTool, + ccrInspectInput, + ccrInspectOutput, + ccrInspectTool, + ccrListInput, + ccrListOutput, + ccrListTool, + ccrDeleteInput, + ccrDeleteOutput, + ccrDeleteTool, + ccrStatsInput, + ccrStatsOutput, + ccrStatsTool, } from "./tools.ts"; // A2A schemas diff --git a/open-sse/mcp-server/schemas/tools.ts b/open-sse/mcp-server/schemas/tools.ts index c6fc9908225..bf77b115779 100644 --- a/open-sse/mcp-server/schemas/tools.ts +++ b/open-sse/mcp-server/schemas/tools.ts @@ -12,6 +12,7 @@ import { z } from "zod"; import { toolSearchTool } from "./toolSearch.ts"; import { pickFastestModelTool } from "./pickFastestModel.ts"; +import { CCR_MCP_TOOLS } from "./ccrTools.ts"; import { AUTO_ROUTING_STRATEGY_VALUES, ROUTING_STRATEGY_VALUES, @@ -24,6 +25,7 @@ import { export type { AuditLevel, McpToolDefinition } from "./toolDefinition.ts"; import type { McpToolDefinition } from "./toolDefinition.ts"; export { pickFastestModelInput, pickFastestModelOutput } from "./pickFastestModel.ts"; +export * from "./ccrTools.ts"; // ============ Phase 1: Essential Tools (8) ============ @@ -1462,6 +1464,7 @@ export const MCP_TOOLS = [ setCompressionEngineTool, listCompressionCombosTool, compressionComboStatsTool, + ...CCR_MCP_TOOLS, oneproxyFetchTool, oneproxyRotateTool, oneproxyStatsTool, diff --git a/open-sse/mcp-server/server.ts b/open-sse/mcp-server/server.ts index 4d8b7be2de9..477fb9d816e 100644 --- a/open-sse/mcp-server/server.ts +++ b/open-sse/mcp-server/server.ts @@ -110,6 +110,7 @@ const TOTAL_MCP_TOOL_COUNT = countUniqueMcpTools({ pluginTools, notionTools, obsidianTools, + compressionTools, }); type JsonRecord = Record; diff --git a/open-sse/mcp-server/toolSearch/catalog.ts b/open-sse/mcp-server/toolSearch/catalog.ts index bd597a4bff5..5cbd39f986b 100644 --- a/open-sse/mcp-server/toolSearch/catalog.ts +++ b/open-sse/mcp-server/toolSearch/catalog.ts @@ -76,9 +76,8 @@ export function getAllToolDefinitions(): ToolCatalogEntry[] { pluginTools, notionTools, obsidianTools, - // compressionTools holds omniroute_ccr_retrieve, which is NOT in MCP_TOOLS — without it - // a `tool_search("compression")` would miss that tool. The other 5 overlap MCP_TOOLS and - // are resolved by the dedup-by-name below (first wins). + // Keep the concrete handler collection in the catalog as a parity guard. Canonical CCR + // definitions now live in MCP_TOOLS too; deduplication below keeps each name visible once. compressionTools, ]; diff --git a/open-sse/mcp-server/tools/compressionTools.ts b/open-sse/mcp-server/tools/compressionTools.ts index 4d1df0f4598..d9e39ffd722 100644 --- a/open-sse/mcp-server/tools/compressionTools.ts +++ b/open-sse/mcp-server/tools/compressionTools.ts @@ -4,6 +4,7 @@ * Tools: * 1. omniroute_compression_status — Get compression config, analytics, and cache stats * 2. omniroute_compression_configure — Update compression settings + * 3. CCR lifecycle tools — Store, retrieve, inspect, list, delete, and stats */ import { logToolCall } from "../audit.ts"; @@ -241,8 +242,23 @@ import { setCompressionEngineInput, listCompressionCombosInput, compressionComboStatsInput, + ccrStoreInput, + ccrRetrieveInput, + ccrInspectInput, + ccrListInput, + ccrDeleteInput, + ccrStatsInput, } from "../schemas/tools.ts"; -import { handleCcrRetrieve } from "../../services/compression/engines/ccr/index.ts"; +import { + MAX_CCR_MCP_FULL_BYTES, + buildCcrReference, + deleteCcrBlock, + getCcrStoreStats, + handleCcrRetrieve, + inspectCcrBlock, + listCcrBlocks, + tryStoreBlock, +} from "../../services/compression/engines/ccr/index.ts"; import { listRtkCommandSamples, discoverRepeatedNoise, @@ -252,22 +268,161 @@ import { import { resolveCallerScopeContext } from "../scopeEnforcement.ts"; import { resolveMcpCallerApiKeyId } from "../mcpCallerIdentity.ts"; -const ccrRetrieveInput = z.object({ - hash: z - .string() - .min(6) - .max(64) - .describe("24-hex content hash from a [CCR retrieve hash=] marker"), - mode: z - .enum(["full", "head", "tail", "lines", "grep", "stats"]) - .optional() - .describe("Retrieval mode: full (default) | head | tail | lines | grep | stats"), - n: z.number().int().positive().max(10000).optional().describe("head/tail: number of lines"), - start: z.number().int().positive().optional().describe("lines: 1-indexed inclusive start"), - end: z.number().int().positive().optional().describe("lines: 1-indexed inclusive end"), - pattern: z.string().max(512).optional().describe("grep: regex (validated safe; ReDoS-rejected)"), - unique: z.boolean().optional().describe("grep: dedupe matching lines"), -}); +async function resolveCcrPrincipal( + extra: McpToolExtraLike | undefined, + scopes: readonly string[] +): Promise { + const apiKeyPrincipal = await resolveMcpCallerApiKeyId(); + if (apiKeyPrincipal) return apiKeyPrincipal; + const { callerId } = resolveCallerScopeContext(extra, scopes); + return callerId === "anonymous" ? undefined : callerId; +} + +export function buildCcrStoreAuditInput(args: z.infer) { + return { + bytes: Buffer.byteLength(args.content, "utf8"), + contentType: args.contentType, + ttlSeconds: args.ttlSeconds, + }; +} + +export async function handleCcrStoreTool( + args: z.infer, + extra?: McpToolExtraLike +) { + const start = Date.now(); + const principal = await resolveCcrPrincipal(extra, ["write:compression"]); + const result = tryStoreBlock(args.content, principal, { + contentType: args.contentType, + source: "mcp", + ttlSeconds: args.ttlSeconds, + }); + const auditInput = buildCcrStoreAuditInput(args); + if (!result.stored) { + const output = { stored: false as const, reason: result.reason }; + await logToolCall( + "omniroute_ccr_store", + auditInput, + output, + Date.now() - start, + false, + result.reason + ); + return output; + } + const output = { + stored: true as const, + reference: buildCcrReference(result.hash, result.metadata.chars), + metadata: result.metadata, + }; + await logToolCall("omniroute_ccr_store", auditInput, output, Date.now() - start, true); + return output; +} + +export async function handleCcrRetrieveTool( + args: z.infer, + extra?: McpToolExtraLike +) { + const start = Date.now(); + const principal = await resolveCcrPrincipal(extra, ["read:compression"]); + const metadata = inspectCcrBlock(args.hash, principal); + if (!metadata) { + const output = { found: false as const, error: "CCR block not found or expired" }; + await logToolCall( + "omniroute_ccr_retrieve", + args, + output, + Date.now() - start, + false, + "NOT_FOUND" + ); + return output; + } + if ((!args.mode || args.mode === "full") && metadata.bytes > MAX_CCR_MCP_FULL_BYTES) { + const output = { + found: true as const, + tooLargeForFull: true as const, + metadata, + suggestedModes: ["head", "tail", "lines", "grep", "stats"] as const, + }; + await logToolCall("omniroute_ccr_retrieve", args, output, Date.now() - start, true); + return output; + } + const queried = handleCcrRetrieve(args, principal); + const refreshedMetadata = inspectCcrBlock(args.hash, principal) ?? metadata; + const output = + "content" in queried + ? { found: true as const, metadata: refreshedMetadata, content: queried.content } + : { found: true as const, metadata: refreshedMetadata, error: queried.error }; + await logToolCall( + "omniroute_ccr_retrieve", + args, + { + ...output, + ...(typeof output.content === "string" + ? { content: `[${Buffer.byteLength(output.content, "utf8")} bytes]` } + : {}), + }, + Date.now() - start, + !("error" in output), + "error" in output ? "INVALID_QUERY" : undefined + ); + return output; +} + +export async function handleCcrInspectTool( + args: z.infer, + extra?: McpToolExtraLike +) { + const start = Date.now(); + const principal = await resolveCcrPrincipal(extra, ["read:compression"]); + const metadata = inspectCcrBlock(args.hash, principal); + const output = metadata + ? { found: true as const, reference: buildCcrReference(args.hash, metadata.chars), metadata } + : { found: false as const }; + await logToolCall("omniroute_ccr_inspect", args, output, Date.now() - start, Boolean(metadata)); + return output; +} + +export async function handleCcrListTool( + args: z.infer, + extra?: McpToolExtraLike +) { + const start = Date.now(); + const principal = await resolveCcrPrincipal(extra, ["read:compression"]); + const result = listCcrBlocks(principal, args); + const output = { + ...result, + entries: result.entries.map((metadata) => ({ + reference: buildCcrReference(metadata.hash, metadata.chars), + metadata, + })), + }; + await logToolCall("omniroute_ccr_list", args, output, Date.now() - start, true); + return output; +} + +export async function handleCcrDeleteTool( + args: z.infer, + extra?: McpToolExtraLike +) { + const start = Date.now(); + const principal = await resolveCcrPrincipal(extra, ["write:compression"]); + const output = { deleted: deleteCcrBlock(args.hash, principal) }; + await logToolCall("omniroute_ccr_delete", args, output, Date.now() - start, true); + return output; +} + +export async function handleCcrStatsTool( + args: z.infer, + extra?: McpToolExtraLike +) { + const start = Date.now(); + const principal = await resolveCcrPrincipal(extra, ["read:compression"]); + const output = getCcrStoreStats(principal); + await logToolCall("omniroute_ccr_stats", args, output, Date.now() - start, true); + return output; +} export async function handleSetCompressionEngine( args: z.infer @@ -323,12 +478,24 @@ export async function handleCompressionComboStats( // T07 — RTK learn/discover exposed via MCP (read-only; suggestions only). Mines the opt-in // raw-output sample store, exactly like the /api/context/rtk/{discover,learn} routes. const rtkDiscoverInput = z.object({ - limit: z.number().int().positive().max(2000).optional().describe("Max samples to scan (default 500)"), + limit: z + .number() + .int() + .positive() + .max(2000) + .optional() + .describe("Max samples to scan (default 500)"), }); const rtkLearnInput = z.object({ command: z.string().min(1).max(500).describe("The command to learn an RTK filter draft for"), - limit: z.number().int().positive().max(2000).optional().describe("Max samples to scan (default 500)"), + limit: z + .number() + .int() + .positive() + .max(2000) + .optional() + .describe("Max samples to scan (default 500)"), }); function resolveSampleLimit(limit?: number): number { @@ -401,6 +568,14 @@ export const compressionTools = { handler: (args: z.infer) => handleCompressionComboStats(args), }, + omniroute_ccr_store: { + name: "omniroute_ccr_store", + description: + "Store verbatim content in the caller-isolated in-memory CCR store and return a ccr:// reference plus the compatible CCR marker. Entries expire automatically and are not persisted across restarts.", + scopes: ["write:compression"], + inputSchema: ccrStoreInput, + handler: handleCcrStoreTool, + }, omniroute_ccr_retrieve: { name: "omniroute_ccr_retrieve", description: @@ -411,22 +586,36 @@ export const compressionTools = { "Scope: read:compression. Always available (sticky-on).", scopes: ["read:compression"], inputSchema: ccrRetrieveInput, - handler: async (args: z.infer, extra?: McpToolExtraLike) => { - // Retrieve must use the SAME principal the CCR store used at compression time: - // `String(apiKeyInfo.id)` (chatCore → getApiKeyMetadata(rawKey)). On MCP HTTP - // transports the raw key lives in httpAuthContext (not in extra.authInfo, since - // OmniRoute auth is API-key not OAuth-clientId) — resolve it to the same key id - // so the block is found. Without this the caller resolved to "anonymous" and the - // store-key never matched (#5649). Cross-tenant IDOR stays closed: a different - // key → different id → miss; no key → undefined → anonymous bucket only. - const apiKeyPrincipal = await resolveMcpCallerApiKeyId(); - if (apiKeyPrincipal) { - return handleCcrRetrieve(args, apiKeyPrincipal); - } - // Fallback (unchanged): OAuth clientId / session scope context, then anonymous. - const { callerId } = resolveCallerScopeContext(extra, ["read:compression"]); - return handleCcrRetrieve(args, callerId === "anonymous" ? undefined : callerId); - }, + handler: handleCcrRetrieveTool, + }, + omniroute_ccr_inspect: { + name: "omniroute_ccr_inspect", + description: "Inspect metadata for a caller-owned CCR block without returning its content.", + scopes: ["read:compression"], + inputSchema: ccrInspectInput, + handler: handleCcrInspectTool, + }, + omniroute_ccr_list: { + name: "omniroute_ccr_list", + description: "List paginated metadata for CCR blocks owned by the current caller.", + scopes: ["read:compression"], + inputSchema: ccrListInput, + handler: handleCcrListTool, + }, + omniroute_ccr_delete: { + name: "omniroute_ccr_delete", + description: "Delete a caller-owned block from the in-memory CCR store.", + scopes: ["write:compression"], + inputSchema: ccrDeleteInput, + handler: handleCcrDeleteTool, + }, + omniroute_ccr_stats: { + name: "omniroute_ccr_stats", + description: + "Return caller-scoped CCR entry and byte usage, lifecycle counters, and in-memory store limits.", + scopes: ["read:compression"], + inputSchema: ccrStatsInput, + handler: handleCcrStatsTool, }, omniroute_rtk_discover: { name: "omniroute_rtk_discover", diff --git a/open-sse/package.json b/open-sse/package.json index c53d7d28fdd..6f38dfd26e4 100644 --- a/open-sse/package.json +++ b/open-sse/package.json @@ -12,6 +12,7 @@ }, "dependencies": { "@toon-format/toon": "^2.3.0", - "safe-regex": "^2.1.1" + "safe-regex": "^2.1.1", + "smol-toml": "1.6.1" } } diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index 10347ec0f3f..bd1dbabe661 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -44,6 +44,8 @@ import { buildSessionQuotaFallback, } from "./quotaTextCooldowns.ts"; import { parseDayGranularityResetMs, shouldPreserveQuotaSignals } from "./quotaResetParsing.ts"; +import { evictLockoutOverflow } from "./accountFallback/lockoutEviction.ts"; +export { MODEL_LOCKOUT_EVICTION_CAP } from "./accountFallback/lockoutEviction.ts"; export type ProviderProfile = { baseCooldownMs: number; @@ -69,7 +71,7 @@ export type ProviderProfile = { }; type JsonRecord = Record; type RateLimitReasonValue = (typeof RateLimitReason)[keyof typeof RateLimitReason]; -type ModelLockoutEntry = { +export type ModelLockoutEntry = { reason: string; until: number; lockedAt: number; @@ -77,7 +79,7 @@ type ModelLockoutEntry = { lastFailureAt: number; resetAfterMs: number; }; -type ModelFailureState = { +export type ModelFailureState = { failureCount: number; lastFailureAt: number; resetAfterMs: number; @@ -467,6 +469,7 @@ function ensureCleanupTimer() { const now = Date.now(); for (const key of modelLockouts.keys()) cleanupModelLockKey(key, now); for (const key of modelFailureState.keys()) cleanupModelLockKey(key, now); + evictModelLockoutOverflow(); }, 15_000); if (typeof _cleanupTimer === "object" && "unref" in _cleanupTimer) { (_cleanupTimer as { unref?: () => void }).unref?.(); // Don't prevent process exit (Node.js only) @@ -476,6 +479,14 @@ function ensureCleanupTimer() { } } +/** @internal exported for testing only (both accessors below). */ +export function evictModelLockoutOverflow(): void { + evictLockoutOverflow(modelLockouts, modelFailureState); +} +export function getModelLockoutSize(): number { + return modelLockouts.size; +} + /** * Lock a specific model on a specific account * @param {string} provider diff --git a/open-sse/services/accountFallback/lockoutEviction.ts b/open-sse/services/accountFallback/lockoutEviction.ts new file mode 100644 index 00000000000..83199830a2b --- /dev/null +++ b/open-sse/services/accountFallback/lockoutEviction.ts @@ -0,0 +1,53 @@ +/** + * accountFallback/lockoutEviction.ts — model-lockout map eviction (cap enforcement). + * + * Extracted from services/accountFallback.ts (file-size gate, #6923): the bounded-growth + * eviction sweep for the in-memory `modelLockouts` / `modelFailureState` maps. Pure w.r.t. + * module state — operates only on the maps passed in by the caller — so it is independently + * testable and reusable outside accountFallback.ts, which wraps evictLockoutOverflow() with + * its own private map instances and re-exports MODEL_LOCKOUT_EVICTION_CAP. + */ + +import type { ModelLockoutEntry, ModelFailureState } from "../accountFallback.ts"; + +// Cap prevents unbounded growth under sustained load. Entries beyond this limit +// are evicted (oldest first, in insertion order) during the periodic cleanup. +export const MODEL_LOCKOUT_EVICTION_CAP = 1000; + +/** + * Evict oldest (insertion-order) entries once a map exceeds the cap — but NEVER a + * still-active (until > now) lockout: cleanupModelLockKey() has already run on every + * key this tick, so anything active left here is a real, in-progress cooldown, and + * dropping it would wrongly let routing resume to it. If a map is still over cap + * purely from active entries, the cap is a soft bound in that rare case rather than + * a correctness trade-off. + */ +export function evictLockoutOverflow( + modelLockouts: Map, + modelFailureState: Map, + cap: number = MODEL_LOCKOUT_EVICTION_CAP +): void { + if (modelLockouts.size > cap) { + const overflow = modelLockouts.size - cap; + const now = Date.now(); + // Only expired entries are eviction candidates (oldest-first, up to the + // overflow count) — active ones never appear in this list at all. + const evictableKeys = [...modelLockouts.entries()] + .filter(([, entry]) => entry.until <= now) + .slice(0, overflow) + .map(([key]) => key); + for (const key of evictableKeys) { + modelLockouts.delete(key); + modelFailureState.delete(key); + } + } + if (modelFailureState.size > cap) { + const overflow = modelFailureState.size - cap; + let evicted = 0; + for (const key of modelFailureState.keys()) { + if (evicted >= overflow) break; + if (!modelLockouts.has(key)) modelFailureState.delete(key); + evicted++; + } + } +} diff --git a/open-sse/services/claudeCodeCompatible.ts b/open-sse/services/claudeCodeCompatible.ts index 01b021579fc..a33e4cd09ed 100644 --- a/open-sse/services/claudeCodeCompatible.ts +++ b/open-sse/services/claudeCodeCompatible.ts @@ -56,6 +56,7 @@ const CLAUDE_CODE_COMPATIBLE_DEFAULT_SYSTEM_BLOCKS = [ const CONTEXT_1M_SUPPORTED_MODELS = [ "claude-fable-5", "claude-sonnet-5", + "claude-sonnet-4-6", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 3eb11d3183f..375e47fb9b3 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -67,6 +67,7 @@ import { estimateTokens } from "./contextManager.ts"; import { getSessionConnection } from "./sessionManager.ts"; import { applySessionStickiness, + normalizeStickinessMessages, recordStickyBinding, clearStickyBinding, peekStickyConnectionId, @@ -137,6 +138,8 @@ import { clampComboDepth, shouldSkipForPredictedTtft, shouldRecordProviderBreakerFailure, + isRequestScopedUpstreamFailure, + shouldSkipConnDisable, resolveDelayMs, comboModelNotFoundResponse, isStreamReadinessFailureErrorBody, @@ -162,6 +165,7 @@ import { resolveWeightedTargets, resolveWeightedStepGroups, } from "./combo/comboStructure.ts"; +import { getKnownContextOverflow } from "./combo/knownContextOverflow.ts"; import { QUOTA_SOFT_DEPRIORITIZE_FACTOR, setCandidateQuotaSoftPenalty, @@ -199,10 +203,21 @@ export { QUOTA_SOFT_DEPRIORITIZE_FACTOR, setCandidateQuotaSoftPenalty }; export { scoreAutoTargets, expandAutoComboCandidatePool }; export type { SingleModelTarget, ResolvedComboTarget }; export { validateResponseQuality }; -export { clampComboDepth, shouldSkipForPredictedTtft, shouldRecordProviderBreakerFailure }; +export { + clampComboDepth, + shouldSkipForPredictedTtft, + shouldRecordProviderBreakerFailure, + isRequestScopedUpstreamFailure, + shouldSkipConnDisable, +}; export { resolveShadowTargets, scheduleShadowRouting }; export { preScreenTargets }; -export { resolveComboRuntimeUnits, resolveComboTargets, filterTargetsByRequestCompatibility }; +export { + resolveComboRuntimeUnits, + resolveComboTargets, + filterTargetsByRequestCompatibility, + getKnownContextOverflow, +}; export { getComboFromData, getComboModelsFromData, @@ -1151,6 +1166,31 @@ export async function handleComboChat({ orderedTargets = await applyRequestTagRouting(orderedTargets, body, log); + const knownContextOverflow = getKnownContextOverflow(orderedTargets, body); + if (knownContextOverflow) { + const { requiredContextTokens, maxKnownContextTokens } = knownContextOverflow; + log.warn( + "COMBO", + `Request context exceeds every known target limit (${requiredContextTokens} > ${maxKnownContextTokens} tokens)` + ); + return errorResponseWithComboDiagnostics( + 400, + `Request requires approximately ${requiredContextTokens} tokens, but the largest known context limit in this combo is ${maxKnownContextTokens} tokens. Reduce or compact the request context.`, + { + poolSize: orderedTargets.length, + attempted: 0, + excluded: orderedTargets.map((target) => ({ + provider: target.provider, + model: target.modelStr, + reason: "context_window", + })), + attemptOrder: [], + terminalReason: "context_length_exceeded", + }, + { code: "context_length_exceeded", type: "invalid_request_error" } + ); + } + if (strategy === "weighted") { log.info( "COMBO", @@ -1244,7 +1284,9 @@ export async function handleComboChat({ ? ({ targets: orderedTargets, messageHash: null, stuck: false } as const) : await applySessionStickiness( orderedTargets, - body.messages as Array<{ role?: string; content?: unknown }> + // #7270: normalize both wire shapes (.messages / Responses-API .input) so the + // stickiness key is derivable on the /v1/responses surface, not just Chat Completions. + normalizeStickinessMessages(body as { messages?: unknown; input?: unknown }) ); orderedTargets = _sticky.targets; orderedTargets = orderTargetsByEvalScores(orderedTargets, config.evalRouting, log); @@ -2064,6 +2106,7 @@ export async function handleComboChat({ : undefined, } : undefined; + const requestScopedFailure = isRequestScopedUpstreamFailure(structuredError); const fallbackResult = checkFallbackError( result.status, errorText, @@ -2176,6 +2219,7 @@ export async function handleComboChat({ status: result.status, sameProviderNext, skipProviderBreaker: fallbackResult.skipProviderBreaker, + requestScopedFailure, }) ) { recordProviderFailure(provider, log, targetWithConnection.connectionId, profile); @@ -2201,7 +2245,7 @@ export async function handleComboChat({ // once the model is cooling down, retrying it would waste an upstream // call and extend the cooldown via exponential backoff. let lockoutRecorded = false; - if (provider && rawModel && retry === 0) { + if (provider && rawModel && retry === 0 && !requestScopedFailure) { const mlSettings = resolveModelLockoutSettings(settings); if (mlSettings.enabled && mlSettings.errorCodes.includes(result.status)) { recordModelLockoutFailure( @@ -2244,7 +2288,7 @@ export async function handleComboChat({ if (i > 0) fallbackCount++; // Wire combo failures into the resilience dashboard (model-level lockout) // alongside the provider-level cooldown below — they govern different scopes. - if (provider && rawModel) { + if (provider && rawModel && !requestScopedFailure) { const mlSettings = resolveModelLockoutSettings(settings); if (mlSettings.enabled && mlSettings.errorCodes.includes(result.status)) { recordModelLockoutFailure( @@ -2273,6 +2317,7 @@ export async function handleComboChat({ resilienceSettings.providerCooldown.enabled && provider && provider !== "unknown" && + !requestScopedFailure && !(result.status === 500 && hasPerModelQuota(provider, rawModel)) ) { recordProviderCooldown( @@ -2545,6 +2590,25 @@ async function handleRoundRobinCombo({ ); const tagFilteredTargets = await applyRequestTagRouting(orderedTargets, body, log); const evalRankedTargets = orderTargetsByEvalScores(tagFilteredTargets, config.evalRouting, log); + const knownContextOverflow = getKnownContextOverflow(evalRankedTargets, body); + if (knownContextOverflow) { + return errorResponseWithComboDiagnostics( + 400, + `Request requires approximately ${knownContextOverflow.requiredContextTokens} tokens, but the largest known context limit in this combo is ${knownContextOverflow.maxKnownContextTokens} tokens. Reduce or compact the request context.`, + { + poolSize: evalRankedTargets.length, + attempted: 0, + excluded: evalRankedTargets.map((target) => ({ + provider: target.provider, + model: target.modelStr, + reason: "context_window", + })), + attemptOrder: [], + terminalReason: "context_length_exceeded", + }, + { code: "context_length_exceeded", type: "invalid_request_error" } + ); + } const filteredTargets = filterTargetsByRequestCompatibility( evalRankedTargets, body, @@ -2668,7 +2732,9 @@ async function handleRoundRobinCombo({ ? ({ targets: filteredTargets, messageHash: null, stuck: false } as const) : await applySessionStickiness( filteredTargets, - body?.messages as Array<{ role?: string; content?: unknown }> + // #7270: normalize both wire shapes (.messages / Responses-API .input) so RR + // stickiness engages on the /v1/responses surface, not just Chat Completions. + normalizeStickinessMessages(body as { messages?: unknown; input?: unknown }) ); let rrStartIndex = startIndex; if (_rrSessionSticky.stuck) { @@ -3039,6 +3105,7 @@ async function handleRoundRobinCombo({ : undefined, } : undefined; + const requestScopedFailure = isRequestScopedUpstreamFailure(structuredError); const fallbackResult = checkFallbackError( result.status, errorText, @@ -3090,6 +3157,7 @@ async function handleRoundRobinCombo({ if ( !isStreamReadinessFailure && !isTokenLimitBreach && + !requestScopedFailure && TRANSIENT_FOR_SEMAPHORE.includes(result.status) && cooldownMs > 0 ) { @@ -3132,6 +3200,7 @@ async function handleRoundRobinCombo({ resilienceSettings.providerCooldown.enabled && provider && provider !== "unknown" && + !requestScopedFailure && !( result.status === 500 && hasPerModelQuota(provider, parseModel(modelStr).model || modelStr) diff --git a/open-sse/services/combo/autoConfig.ts b/open-sse/services/combo/autoConfig.ts index 2815444f256..ed56f11989e 100644 --- a/open-sse/services/combo/autoConfig.ts +++ b/open-sse/services/combo/autoConfig.ts @@ -1,4 +1,5 @@ import { DEFAULT_WEIGHTS, type ScoringWeights } from "../autoCombo/scoring.ts"; +import { getModePack } from "../autoCombo/modePacks.ts"; import { isRecord } from "./comboData.ts"; import { resolveResetWindowConfig, resolveSlaRoutingPolicy } from "./quotaScoring.ts"; import type { ComboLike, ResolvedComboTarget } from "./types.ts"; @@ -34,7 +35,7 @@ export function parseAutoConfig(combo: ComboLike, eligibleTargets: ResolvedCombo ? autoConfigSource.candidatePool : [...new Set(eligibleTargets.map((target) => target.provider))]; - const weights = + const configuredWeights = autoConfigSource.weights && typeof autoConfigSource.weights === "object" ? (autoConfigSource.weights as ScoringWeights) : DEFAULT_WEIGHTS; @@ -52,6 +53,7 @@ export function parseAutoConfig(combo: ComboLike, eligibleTargets: ResolvedCombo : undefined; const modePack = typeof autoConfigSource.modePack === "string" ? autoConfigSource.modePack : undefined; + const weights = modePack ? getModePack(modePack) || configuredWeights : configuredWeights; const resetWindowConfig = resolveResetWindowConfig(autoConfigSource); const slaPolicy = resolveSlaRoutingPolicy(autoConfigSource); diff --git a/open-sse/services/combo/comboPredicates.ts b/open-sse/services/combo/comboPredicates.ts index 887709522fa..39ba927640c 100644 --- a/open-sse/services/combo/comboPredicates.ts +++ b/open-sse/services/combo/comboPredicates.ts @@ -8,6 +8,7 @@ import { errorResponse } from "../../utils/error.ts"; import { parseModel } from "../model.ts"; +import { isSelfInflictedUpstreamTimeout } from "../../handlers/chatCore/cooldownClassification.ts"; import type { ResolvedComboTarget } from "./types.ts"; // Status codes that should mark round-robin target semaphores as cooling down. @@ -150,12 +151,53 @@ export function shouldRecordProviderBreakerFailure(args: { status: number; sameProviderNext: boolean; skipProviderBreaker?: boolean; + requestScopedFailure?: boolean; }): boolean { return ( !args.isStreamReadinessFailure && PROVIDER_BREAKER_FAILURE_STATUSES.has(args.status) && !args.sameProviderNext && - !args.skipProviderBreaker + !args.skipProviderBreaker && + !args.requestScopedFailure + ); +} + +const REQUEST_SCOPED_UPSTREAM_ERROR_CODES = new Set([ + "context_length_exceeded", + "upstream_empty_response", + "upstream_response_failed", +]); + +/** Request/model-specific failures must not poison provider-wide resilience state. */ +export function isRequestScopedUpstreamFailure(error?: { + code?: string | null; + type?: string | null; +}): boolean { + const code = typeof error?.code === "string" ? error.code.toLowerCase() : ""; + const type = typeof error?.type === "string" ? error.type.toLowerCase() : ""; + return REQUEST_SCOPED_UPSTREAM_ERROR_CODES.has(code) || type === "context_length_exceeded"; +} + +/** + * #7177: whether handleSingleModelChat should skip the connection-level cooldown + * (markAccountUnavailable) for a failed attempt — client disconnects, a 401 when the + * connection has extra keys to rotate through, a known request-scoped upstream failure + * (e.g. context overflow — not a connection health signal), or our own self-inflicted + * timeout all mean the connection itself is healthy and should not be cooled down. + */ +export function shouldSkipConnDisable( + result: { status: number; errorCode?: string | null; errorType?: string | null }, + is401: boolean, + hasExtraKeys: boolean, + provider: string +): boolean { + return ( + result.status === 499 || + result.errorCode === "client_disconnected" || + result.errorType === "client_disconnected" || + (is401 && hasExtraKeys) || + isRequestScopedUpstreamFailure({ code: result.errorCode, type: result.errorType }) || + isSelfInflictedUpstreamTimeout(result.status, result.errorType, provider) ); } diff --git a/open-sse/services/combo/comboStructure.ts b/open-sse/services/combo/comboStructure.ts index a98c8e820ba..0b7f8cb0fa4 100644 --- a/open-sse/services/combo/comboStructure.ts +++ b/open-sse/services/combo/comboStructure.ts @@ -18,6 +18,7 @@ import { getResolvedModelCapabilities } from "../modelCapabilities.ts"; import { parseModel } from "../model.ts"; import { dedupeTargetsByExecutionKey, isRecord } from "./comboData.ts"; import { getTargetProvider, MAX_COMBO_DEPTH } from "./comboPredicates.ts"; +import { hasEstimableContent } from "./knownContextOverflow.ts"; import { normalizeModelEntry, orderTargetsForWeightedFallback, @@ -409,7 +410,7 @@ export function getModelContextLimitForModelString(modelStr: string) { return getModelContextLimit(provider, model); } -type RequestCompatibilityRequirements = { +export type RequestCompatibilityRequirements = { requiresTools: boolean; requiresVision: boolean; requiresStructuredOutput: boolean; @@ -438,7 +439,7 @@ function requestRequiresStructuredOutput(body: Record): boolean function estimateRequestInputTokens(body: Record): number { const estimatePayload: Record = {}; for (const key of ["messages", "input", "tools", "functions", "response_format"]) { - if (body[key] !== undefined) estimatePayload[key] = body[key]; + if (hasEstimableContent(body[key])) estimatePayload[key] = body[key]; } return Object.keys(estimatePayload).length > 0 ? estimateTokens(estimatePayload) : 0; } @@ -460,7 +461,7 @@ function valueContainsImagePart(value: unknown, depth = 0): boolean { return Object.values(value).some((entry) => valueContainsImagePart(entry, depth + 1)); } -function deriveRequestCompatibilityRequirements( +export function deriveRequestCompatibilityRequirements( body: Record ): RequestCompatibilityRequirements { const estimatedInputTokens = estimateRequestInputTokens(body); @@ -486,21 +487,57 @@ function exceedsKnownOutputLimit( return maxOutputTokens < requestedOutputTokens; } -function getKnownContextLimit(capabilities: { - maxInputTokens?: number | null; - contextWindow?: number | null; -}): number | null { - return capabilities.maxInputTokens ?? capabilities.contextWindow ?? null; +/** + * Decide whether a target's known context limit accommodates the request. + * + * `maxInputTokens` is an **input-only** cap — the requested output reserve is + * already enforced separately against `maxOutputTokens` (see + * `exceedsKnownOutputLimit`), so it must NOT be re-counted here. Comparing + * `maxInputTokens` against `estimatedInputTokens + requestedOutputTokens` + * double-counted the output reserve and shrank the effective input allowance + * (#7039). + * + * `contextWindow` is the total window, so input + output must both fit. + * + * Returns `true` when the known limit accommodates the request, `false` when + * it is known to be too small, and `null` when no limit metadata is known. + */ +function evaluateContextLimit( + capabilities: { maxInputTokens?: number | null; contextWindow?: number | null }, + requirements: { estimatedInputTokens: number; requiredContextTokens: number } +): boolean | null { + const hasMaxInput = capabilities.maxInputTokens != null; + const hasContextWindow = capabilities.contextWindow != null; + + // Neither limit is known — cannot judge. + if (!hasMaxInput && !hasContextWindow) return null; + + // The input-only cap must accommodate the estimated input. + const inputFits = hasMaxInput + ? capabilities.maxInputTokens! >= requirements.estimatedInputTokens + : true; + + // The total window must accommodate input + requested output. The output + // reserve is enforced separately via `maxOutputTokens`, but when a model + // exposes both `maxInputTokens` and `contextWindow` the two must not be + // checked in isolation: a request whose input fits `maxInputTokens` but whose + // input + output exceeds `contextWindow` must still be rejected (#7039 + // follow-up — shared-window models where `maxInputTokens` defaults to the + // total window size). + const totalFits = hasContextWindow + ? capabilities.contextWindow! >= requirements.requiredContextTokens + : true; + + return inputFits && totalFits; } function hasKnownCompatibleContextLimit( target: ResolvedComboTarget, - requiredContextTokens: number + requirements: RequestCompatibilityRequirements ): boolean { - if (requiredContextTokens <= 0) return false; + if (requirements.requiredContextTokens <= 0) return false; const capabilities = getResolvedModelCapabilities(target.modelStr); - const contextLimit = getKnownContextLimit(capabilities); - return contextLimit !== null && contextLimit >= requiredContextTokens; + return evaluateContextLimit(capabilities, requirements) === true; } function hasOnlyContextWindowFailures(reasons: string[]): boolean { @@ -539,12 +576,8 @@ function getTargetCompatibilityFailures( failures.push("output_tokens"); } - const contextLimit = getKnownContextLimit(capabilities); - if ( - requirements.requiredContextTokens > 0 && - contextLimit !== null && - contextLimit < requirements.requiredContextTokens - ) { + const contextVerdict = evaluateContextLimit(capabilities, requirements); + if (requirements.requiredContextTokens > 0 && contextVerdict === false) { failures.push("context_window"); } @@ -584,7 +617,7 @@ export function filterTargetsByRequestCompatibility( ); if (requirements.requiredContextTokens > 0 && rejectedForContextWindow) { const knownContextCompatible = compatible.filter((target) => - hasKnownCompatibleContextLimit(target, requirements.requiredContextTokens) + hasKnownCompatibleContextLimit(target, requirements) ); if (knownContextCompatible.length > 0 && knownContextCompatible.length < compatible.length) { diff --git a/open-sse/services/combo/knownContextOverflow.ts b/open-sse/services/combo/knownContextOverflow.ts new file mode 100644 index 00000000000..4416d1d4516 --- /dev/null +++ b/open-sse/services/combo/knownContextOverflow.ts @@ -0,0 +1,97 @@ +/** + * Known context-overflow rejection, extracted from comboStructure.ts to keep + * that file under the file-size cap (#7177). + * + * Fixes: routing a request to a combo whose targets all have a KNOWN (not + * unknown/fail-open) context window too small for the request used to be + * discovered only after every target was tried and failed upstream — burning + * retries/cooldowns on a request that could never succeed. This lets the + * combo dispatcher reject it up front, before exhausting providers. + * + * getKnownContextLimit/hasEstimableContent also + * live here (moved from comboStructure.ts, same file-size-cap motivation): + * they are the "how big is a target's known context window" primitives, so + * they belong next to the overflow check that is their main consumer. + * comboStructure.ts's own compatibility filter now decides fit via its + * evaluateContextLimit (#7052); only hasEstimableContent is imported back. + */ + +import { getResolvedModelCapabilities } from "../modelCapabilities.ts"; +import { deriveRequestCompatibilityRequirements } from "./comboStructure.ts"; +import type { ResolvedComboTarget } from "./types.ts"; + +export type KnownContextOverflow = { + estimatedInputTokens: number; + requestedOutputTokens: number; + requiredContextTokens: number; + maxKnownContextTokens: number; + targetCount: number; +}; + +// #7177: an empty array/object (e.g. a default `messages: []` some combo entrypoints inject +// when the caller sent none) has no real content — counting it would charge a few phantom +// "structural" tokens (JSON.stringify braces/brackets) toward the estimate, which is enough +// to falsely trip the exact-boundary known-context-overflow check for a request that has no +// actual input at all. +export function hasEstimableContent(value: unknown): boolean { + if (value === undefined || value === null) return false; + if (Array.isArray(value)) return value.length > 0; + if (typeof value === "object") return Object.keys(value).length > 0; + return true; +} + +// #7177: known context limit that accounts for the request's own requested +// output tokens — a target whose input+output would together exceed +// maxInputTokens is exactly as incompatible as one whose contextWindow is too +// small, so both bounds go through the same min() so far the tightest wins. +export function getKnownContextLimit( + capabilities: { + maxInputTokens?: number | null; + contextWindow?: number | null; + }, + requestedOutputTokens = 0 +): number | null { + const limits: number[] = []; + if (capabilities.maxInputTokens != null) { + limits.push(capabilities.maxInputTokens + requestedOutputTokens); + } + if (capabilities.contextWindow != null) { + limits.push(capabilities.contextWindow); + } + return limits.length > 0 ? Math.min(...limits) : null; +} + + +/** + * Return a hard context-overflow decision only when every target has a known + * context limit and every one of those limits is too small for the request. + * Unknown metadata deliberately keeps the legacy fail-open behavior. + */ +export function getKnownContextOverflow( + targets: ResolvedComboTarget[], + body: Record +): KnownContextOverflow | null { + if (targets.length === 0) return null; + const requirements = deriveRequestCompatibilityRequirements(body); + if (requirements.requiredContextTokens <= 0) return null; + + const limits = targets.map((target) => + getKnownContextLimit( + getResolvedModelCapabilities(target.modelStr), + requirements.requestedOutputTokens + ) + ); + if (limits.some((limit) => limit === null)) return null; + + const knownLimits = limits as number[]; + const maxKnownContextTokens = Math.max(...knownLimits); + if (maxKnownContextTokens >= requirements.requiredContextTokens) return null; + + return { + estimatedInputTokens: requirements.estimatedInputTokens, + requestedOutputTokens: requirements.requestedOutputTokens, + requiredContextTokens: requirements.requiredContextTokens, + maxKnownContextTokens, + targetCount: targets.length, + }; +} diff --git a/open-sse/services/combo/resolveAutoStrategy.ts b/open-sse/services/combo/resolveAutoStrategy.ts index ef6082f9af3..9d38801b35c 100644 --- a/open-sse/services/combo/resolveAutoStrategy.ts +++ b/open-sse/services/combo/resolveAutoStrategy.ts @@ -10,6 +10,7 @@ import { } from "../autoCombo/requestControls.ts"; import { selectWithStrategy } from "../autoCombo/routerStrategy.ts"; import { buildComplexityRoutingHint } from "../autoCombo/complexityRouter"; +import { getModePack } from "../autoCombo/modePacks.ts"; import { recordComboIntent } from "../comboMetrics.ts"; import { estimateTokens } from "../contextManager.ts"; import { classifyWithConfig } from "../intentClassifier.ts"; @@ -160,7 +161,7 @@ export async function resolveAutoStrategyOrder( const { routingStrategy, candidatePool, - weights, + weights: configWeights, explorationRate, budgetCap: configBudgetCap, budgetFallback: configBudgetFallback, @@ -180,6 +181,17 @@ export async function resolveAutoStrategyOrder( const budgetFallback = requestBudgetFallback ?? configBudgetFallback; const requestModePack = resolveRequestModePack(relayOptions?.mode); const modePack = requestModePack.override ? requestModePack.modePack : configModePack; + // #7008: `weights` must track the *effective* (post-override) modePack, not just + // the combo's stored one. `selectAutoProvider()` (engine.ts) already re-derives + // weights internally from the `modePack` it's given, so it correctly reacts to a + // per-request X-OmniRoute-Mode override — but `scoreAutoTargets()` (the fallback + // ranking below) has no such re-derivation and only ever sees whatever `weights` + // it's handed. Without this recompute, a request overriding e.g. `quality-first` + // to `ship-fast` would select its primary target under ship-fast weights but rank + // every fallback under the stale quality-first weights — the same + // select-under-one-policy/rank-under-another bug this module's original fix + // (parseAutoConfig honoring the combo's own stored modePack) set out to close. + const weights = modePack ? getModePack(modePack) || configWeights : configWeights; if (requestModePack.override || requestBudgetCap !== undefined || requestBudgetFallback !== undefined) { log.debug?.( "COMBO", diff --git a/open-sse/services/combo/sessionStickiness.ts b/open-sse/services/combo/sessionStickiness.ts index d6d6eb06dd5..50bb9c796d7 100644 --- a/open-sse/services/combo/sessionStickiness.ts +++ b/open-sse/services/combo/sessionStickiness.ts @@ -254,6 +254,40 @@ const stickyMap = new Map(); // ─── Helpers ───────────────────────────────────────────────────────────────── +/** + * #7270: Normalize a request body's user turns into a `{role, content}[]` view for + * stickiness-key derivation, covering both wire formats: + * - Chat Completions (`/v1/chat/completions`) → turns live in `.messages`. + * - OpenAI Responses API (`/v1/responses`) → turns live in `.input`, which may be a + * plain string OR an array of message items; `.messages` is never populated. Array + * items may themselves be bare strings (shorthand for a user message) — the same + * shape `responsesInputNormalization.ts`'s `normalizeCodexResponsesInputItem` + * already special-cases — so those are mapped to `{role: "user", content: item}`. + * Combo target ordering runs BEFORE per-target format translation, so without this + * the Responses-API key resolved to null and stickiness silently no-oped for the + * entire surface (round-robin/random/strict-random all re-ordered every turn). + * `.messages` takes precedence when present (Chat Completions), then `.input`. + * Returns null when neither carrier yields turns (fail-open, same as deriveMessageHash). + */ +export function normalizeStickinessMessages( + body: { messages?: unknown; input?: unknown } | null | undefined +): Array<{ role?: string; content?: unknown }> | null { + if (!body || typeof body !== "object") return null; + const { messages, input } = body as { messages?: unknown; input?: unknown }; + if (Array.isArray(messages) && messages.length > 0) { + return messages as Array<{ role?: string; content?: unknown }>; + } + if (typeof input === "string" && input.length > 0) { + return [{ role: "user", content: input }]; + } + if (Array.isArray(input) && input.length > 0) { + return input.map((item) => + typeof item === "string" ? { role: "user", content: item } : item + ) as Array<{ role?: string; content?: unknown }>; + } + return null; +} + /** * Derive a stable 16-hex-char session key from the first user message content. * Returns null when the message cannot be extracted (fail-open). diff --git a/open-sse/services/combo/targetExhaustion.ts b/open-sse/services/combo/targetExhaustion.ts index 7091b483ef1..66f7ef5cd20 100644 --- a/open-sse/services/combo/targetExhaustion.ts +++ b/open-sse/services/combo/targetExhaustion.ts @@ -21,7 +21,7 @@ import { isProviderExhaustedReason, } from "../accountFallback.ts"; import { RateLimitReason } from "../../config/constants.ts"; -import { isProviderCircuitOpenResult } from "./comboPredicates.ts"; +import { isProviderCircuitOpenResult, isRequestScopedUpstreamFailure } from "./comboPredicates.ts"; import type { ComboLogger, ResolvedComboTarget } from "./types.ts"; // Connection-level failure statuses: the provider connection itself is likely bad (upstream @@ -104,7 +104,15 @@ export function applyComboTargetExhaustion( if (result.status === 429 && !isTokenLimitBreach && provider && provider !== "unknown") { transientRateLimitedProviders.add(provider); } - markConnectionLevelExhaustion(target, { result, errorText, sets, log, tag, rawModel }); + markConnectionLevelExhaustion(target, { + result, + errorText, + sets, + log, + tag, + rawModel, + structuredError, + }); } return providerExhausted; @@ -120,16 +128,17 @@ function markConnectionLevelExhaustion( target: ResolvedComboTarget, opts: Pick< ApplyComboTargetExhaustionOptions, - "result" | "errorText" | "sets" | "log" | "tag" | "rawModel" + "result" | "errorText" | "sets" | "log" | "tag" | "rawModel" | "structuredError" > ): void { - const { result, errorText, sets, log, tag, rawModel } = opts; + const { result, errorText, sets, log, tag, rawModel, structuredError } = opts; const provider = target.provider; if ( !provider || provider === "unknown" || !CONNECTION_LEVEL_ERROR_STATUSES.includes(result.status) || isProviderCircuitOpenResult(result, errorText) || + isRequestScopedUpstreamFailure(structuredError) || // #5085: empty-content 502 is a healthy connection returning no body — model-level, not // connection-level. Don't exhaust the provider; let the remaining legs (incl. same-provider) // be tried in-request. diff --git a/open-sse/services/combo/targetSorters.ts b/open-sse/services/combo/targetSorters.ts index 3432b639fe5..7ddd3e45098 100644 --- a/open-sse/services/combo/targetSorters.ts +++ b/open-sse/services/combo/targetSorters.ts @@ -104,41 +104,23 @@ export async function sortTargetsByCost(targets: ResolvedComboTarget[]) { .filter((target): target is ResolvedComboTarget => target !== null); } -/** - * Sort models by usage count (least-used first) for least-used strategy - * @param {Array} models - Model strings - * @param {string} comboName - Combo name for metrics lookup - * @returns {Array} Sorted model strings - */ -export function sortModelsByUsage(models: string[], comboName: string): string[] { +export function sortTargetsByUsage(targets: ResolvedComboTarget[], comboName: string) { const metrics = getComboMetrics(comboName); - if (!metrics?.byModel) return models; - - const withUsage = models.map((modelStr) => ({ - modelStr, - requests: metrics.byModel[modelStr]?.requests ?? 0, - })); + if (!metrics) return targets; + + // Key on executionKey (unique per model + account) so a combo that repeats the + // same modelStr across DISTINCT accounts distributes by per-account usage + // instead of by the shared modelStr. The old code grouped targets under + // modelStr and read byModel[modelStr] (which aggregates every account of that + // model), so all accounts collapsed into one bucket and the first account + // always won — exhausting it while the others stayed idle (#7015). Per-target + // usage lives in byTarget[executionKey]; unknown targets rank as 0. + const withUsage = targets.map((target) => { + const requests = metrics.byTarget?.[target.executionKey]?.requests ?? 0; + return { target, requests }; + }); withUsage.sort((a, b) => a.requests - b.requests); - return withUsage.map((e) => e.modelStr); -} - -export function sortTargetsByUsage(targets: ResolvedComboTarget[], comboName: string) { - const orderedModels = sortModelsByUsage( - targets.map((target) => target.modelStr), - comboName - ); - const byModel = new Map(); - for (const target of targets) { - const queue = byModel.get(target.modelStr) || []; - queue.push(target); - byModel.set(target.modelStr, queue); - } - return orderedModels - .map((modelStr) => { - const queue = byModel.get(modelStr); - return queue?.shift() || null; - }) - .filter((target): target is ResolvedComboTarget => target !== null); + return withUsage.map((e) => e.target); } function getP2CTargetScore( diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index 99805eb35ca..f5330ae7736 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -566,8 +566,30 @@ export async function validateResponseQuality( const reasoningContent = message.reasoning_content ?? message.reasoning; const hasReasoningContent = typeof reasoningContent === "string" && reasoningContent.trim().length > 0; - const hasContent = - (content !== null && content !== undefined && content !== "") || hasReasoningContent; + // Issue #7000: content can be a string, an array of content parts + // (multimodal), or null. An empty array [] or an array of empty parts + // must NOT count as valid content — only arrays with at least one + // non-empty text/image part do. + let hasContent: boolean; + if (Array.isArray(content)) { + hasContent = content.some( + (part) => + !!part && + typeof part === "object" && + ((typeof (part as Record).text === "string" && + ((part as Record).text as string).trim().length > 0) || + (part as Record).type === "image_url" || + (part as Record).type === "input_audio" || + (part as Record).type === "file") + ); + } else { + hasContent = + (content !== null && + content !== undefined && + content !== "" && + (typeof content !== "string" || content.trim().length > 0)) || + hasReasoningContent; + } const hasToolCalls = Array.isArray(toolCalls) && toolCalls.length > 0; if (!hasContent && !hasToolCalls) { diff --git a/open-sse/services/compression/bodyAdapter.ts b/open-sse/services/compression/bodyAdapter.ts index 9c42fbdd6ff..af407c510de 100644 --- a/open-sse/services/compression/bodyAdapter.ts +++ b/open-sse/services/compression/bodyAdapter.ts @@ -14,7 +14,11 @@ type ResponsesItem = { [key: string]: unknown; }; -const RESPONSES_MESSAGE_TYPES = new Set(["message", "function_call_output"]); +const RESPONSES_MESSAGE_TYPES = new Set([ + "message", + "function_call_output", + "custom_tool_call_output", +]); const COMPRESSION_INPUT_INDEX = Symbol("compressionInputIndex"); // Kiro envelope path back to the original tool-result text inside @@ -60,11 +64,47 @@ function fromChatContent(nextContent: unknown, originalContent: unknown): unknow return nextContent; } +function customToolOutputToChatContent(rawOutput: unknown): unknown { + if (typeof rawOutput !== "string") { + if (isRecord(rawOutput) && typeof rawOutput.output === "string") return rawOutput.output; + return rawOutput; + } + + try { + const parsed = JSON.parse(rawOutput) as unknown; + if (isRecord(parsed) && typeof parsed.output === "string") return parsed.output; + } catch { + // Plain-text custom tool output is already in the form compression engines expect. + } + return rawOutput; +} + +function restoreCustomToolOutput(nextContent: unknown, originalOutput: unknown): unknown { + if (typeof originalOutput === "string") { + try { + const parsed = JSON.parse(originalOutput) as unknown; + if (isRecord(parsed) && typeof parsed.output === "string") { + return JSON.stringify({ ...parsed, output: nextContent }); + } + } catch { + // Preserve the original plain-text representation below. + } + } + if (isRecord(originalOutput) && typeof originalOutput.output === "string") { + return { ...originalOutput, output: nextContent }; + } + return fromChatContent(nextContent, originalOutput); +} + +function responsesToolOutputField(item: ResponsesItem): "output" | "content" { + return item.output !== null && item.output !== undefined ? "output" : "content"; +} + function responsesItemToMessage(item: ResponsesItem): MessageLike | null { const type = typeof item.type === "string" ? item.type : "message"; if (!RESPONSES_MESSAGE_TYPES.has(type)) return null; - if (type === "function_call_output") { + if (type === "function_call_output" || type === "custom_tool_call_output") { const rawOutput = item.output ?? item.content; // OpenAI Responses shape (Codex): body.input holds Responses items. When // output is a JSON object (not a string or content array), serialise it so @@ -77,7 +117,12 @@ function responsesItemToMessage(item: ResponsesItem): MessageLike | null { !Array.isArray(rawOutput); return { role: "tool", - content: isObjectOutput ? JSON.stringify(rawOutput) : toChatContent(rawOutput), + content: + type === "custom_tool_call_output" + ? customToolOutputToChatContent(rawOutput) + : isObjectOutput + ? JSON.stringify(rawOutput) + : toChatContent(rawOutput), }; } @@ -89,10 +134,15 @@ function responsesItemToMessage(item: ResponsesItem): MessageLike | null { function messageToResponsesItem(message: MessageLike, originalItem: ResponsesItem): ResponsesItem { const type = typeof originalItem.type === "string" ? originalItem.type : "message"; - if (type === "function_call_output") { + if (type === "function_call_output" || type === "custom_tool_call_output") { + const outputField = responsesToolOutputField(originalItem); + const originalOutput = originalItem[outputField]; return { ...originalItem, - output: fromChatContent(message.content, originalItem.output), + [outputField]: + type === "custom_tool_call_output" + ? restoreCustomToolOutput(message.content, originalOutput) + : fromChatContent(message.content, originalOutput), }; } @@ -332,7 +382,12 @@ function rewriteKiroEntry( let trChanged = false; const nextContent = content.map((part, partIdx) => { if (!isRecord(part) || typeof part.text !== "string") return part; - const key = kiroPathKey({ scope, historyIndex, toolResultIndex: trIdx, contentIndex: partIdx }); + const key = kiroPathKey({ + scope, + historyIndex, + toolResultIndex: trIdx, + contentIndex: partIdx, + }); const rewritten = rewrites.get(key); if (rewritten === undefined || rewritten === part.text) return part; trChanged = true; diff --git a/open-sse/services/compression/engines/ccr/index.ts b/open-sse/services/compression/engines/ccr/index.ts index 3308d201edd..5aee6c282db 100644 --- a/open-sse/services/compression/engines/ccr/index.ts +++ b/open-sse/services/compression/engines/ccr/index.ts @@ -26,8 +26,8 @@ * does not affect another's (cross-tenant state drift protection). * * Memory bound: - * - Both `ccrStore` and `retrievalCounts` are capped at MAX_CCR_ENTRIES - * entries using FIFO eviction (Map insertion-order guarantees). + * - Entries are capped by count, global bytes, per-principal bytes, block bytes and TTL. + * - Expired and least-recently-used entries are removed before a store is rejected. * * Conservative guards: * - Never touch `role: "system"`. @@ -60,12 +60,69 @@ const RETRIEVAL_THRESHOLD = 3; * ramp (only the >= threshold cliff remains — the legacy binary behavior). */ const RETRIEVAL_RAMP_FACTOR_DEFAULT = 2; -/** - * Maximum number of entries in each bounded store. - * When inserting beyond this cap, the oldest entry (Map insertion order) is evicted. - * 5 000 entries × ~2 KB average ≈ 10 MB upper bound for each map. - */ +/** Maximum number of entries in the principal-scoped, LRU-ordered store. */ export const MAX_CCR_ENTRIES = 5_000; +export const MAX_CCR_BLOCK_BYTES = 2 * 1024 * 1024; +export const MAX_CCR_PRINCIPAL_BYTES = 16 * 1024 * 1024; +export const MAX_CCR_GLOBAL_BYTES = 64 * 1024 * 1024; +export const DEFAULT_CCR_TTL_SECONDS = 24 * 60 * 60; +export const MAX_CCR_TTL_SECONDS = 7 * 24 * 60 * 60; +export const MAX_CCR_MCP_FULL_BYTES = 256 * 1024; + +export type CcrEntrySource = "compression" | "mcp" | "ionizer" | "session-dedup"; + +export interface CcrEntryMetadata { + hash: string; + bytes: number; + chars: number; + lines: number; + contentType: string; + source: CcrEntrySource; + createdAt: number; + lastAccessedAt: number; + expiresAt: number; + retrievalCount: number; +} + +type CcrEntry = Omit & { + principalId: string; + content: string; +}; + +export interface StoreCcrBlockOptions { + contentType?: string; + source?: CcrEntrySource; + ttlSeconds?: number; + now?: number; +} + +export type StoreCcrBlockResult = + | { stored: true; hash: string; metadata: CcrEntryMetadata } + | { + stored: false; + hash: string; + reason: "block_too_large" | "principal_budget_exceeded" | "global_budget_exceeded"; + }; + +export interface CcrStoreStats { + storage: "memory"; + entries: number; + bytes: number; + limits: { + maxEntries: number; + maxBlockBytes: number; + maxPrincipalBytes: number; + maxGlobalBytes: number; + defaultTtlSeconds: number; + maxTtlSeconds: number; + maxMcpFullBytes: number; + }; + lifecycle: { + expiredEvictions: number; + capacityEvictions: number; + rejectedStores: number; + }; +} // ─── principal-scoped, bounded content store ────────────────────────────────── @@ -73,9 +130,12 @@ export const MAX_CCR_ENTRIES = 5_000; * Store key = `${principalId ?? "__anon__"} ${contentHash}`. * Using a compound key scopes data to the principal that stored it. */ -const ccrStore = new Map(); -/** Retrieval counter store — same scoping as ccrStore. */ +const ccrStore = new Map(); const retrievalCounts = new Map(); +const principalBytesMap = new Map(); +let ccrTotalBytes = 0; +type CcrLifecycleCounters = CcrStoreStats["lifecycle"]; +const lifecycleByPrincipal = new Map(); /** Sentinel used when no principalId is provided. */ const ANON = "__anon__"; @@ -84,18 +144,88 @@ function buildStoreKey(hash: string, principalId?: string): string { return `${principalId ?? ANON} ${hash}`; } -/** - * Insert a value into a bounded Map, evicting the oldest entry when over the cap. - */ -function boundedSet(map: Map, key: string, value: V): void { - if (!map.has(key) && map.size >= MAX_CCR_ENTRIES) { - // Map preserves insertion order — the first iterator result is the oldest entry. - const firstKey = map.keys().next().value; - if (firstKey !== undefined) { - map.delete(firstKey); +function readLifecycleCounters(principalId: string): CcrLifecycleCounters { + return ( + lifecycleByPrincipal.get(principalId) ?? { + expiredEvictions: 0, + capacityEvictions: 0, + rejectedStores: 0, } + ); +} + +function mutableLifecycleCounters(principalId: string): CcrLifecycleCounters { + const existing = lifecycleByPrincipal.get(principalId); + if (existing) return existing; + const counters = { expiredEvictions: 0, capacityEvictions: 0, rejectedStores: 0 }; + lifecycleByPrincipal.set(principalId, counters); + return counters; +} + +function publicMetadata(entry: CcrEntry): CcrEntryMetadata { + const { principalId: _principalId, content: _content, ...metadata } = entry; + return { + ...metadata, + retrievalCount: retrievalCounts.get(buildStoreKey(entry.hash, entry.principalId)) ?? 0, + }; +} + +function setRetrievalCount(key: string, count: number): void { + if (!retrievalCounts.has(key) && retrievalCounts.size >= MAX_CCR_ENTRIES) { + const oldestKey = retrievalCounts.keys().next().value; + if (oldestKey !== undefined) retrievalCounts.delete(oldestKey); } - map.set(key, value); + retrievalCounts.delete(key); + retrievalCounts.set(key, count); +} + +function removeEntry(key: string, reason?: "expired" | "capacity"): boolean { + const entry = ccrStore.get(key); + if (!entry) return false; + ccrStore.delete(key); + ccrTotalBytes = Math.max(0, ccrTotalBytes - entry.bytes); + const remainingPrincipalBytes = Math.max( + 0, + (principalBytesMap.get(entry.principalId) ?? 0) - entry.bytes + ); + if (remainingPrincipalBytes === 0) principalBytesMap.delete(entry.principalId); + else principalBytesMap.set(entry.principalId, remainingPrincipalBytes); + const counters = mutableLifecycleCounters(entry.principalId); + if (reason === "expired") counters.expiredEvictions++; + if (reason === "capacity") counters.capacityEvictions++; + return true; +} + +function purgeExpired(now = Date.now()): void { + for (const [key, entry] of ccrStore) { + if (entry.expiresAt <= now) removeEntry(key, "expired"); + } +} + +function getActiveEntry(key: string, now = Date.now()): CcrEntry | null { + const entry = ccrStore.get(key); + if (!entry) return null; + if (entry.expiresAt <= now) { + removeEntry(key, "expired"); + return null; + } + return entry; +} + +function principalBytes(principalId: string): number { + return principalBytesMap.get(principalId) ?? 0; +} + +function evictOldestMatching(predicate: (entry: CcrEntry) => boolean): boolean { + for (const [key, entry] of ccrStore) { + if (predicate(entry)) return removeEntry(key, "capacity"); + } + return false; +} + +function normalizeTtlSeconds(value: number | undefined): number { + if (!Number.isFinite(value) || value === undefined) return DEFAULT_CCR_TTL_SECONDS; + return Math.max(60, Math.min(MAX_CCR_TTL_SECONDS, Math.floor(value))); } /** @@ -107,26 +237,114 @@ function hashContent(text: string): string { return crypto.createHash("sha256").update(text).digest("hex").slice(0, 24); } +function rejectStore( + hash: string, + owner: string, + reason: Exclude["reason"] +): StoreCcrBlockResult { + mutableLifecycleCounters(owner).rejectedStores++; + return { stored: false, hash, reason }; +} + +function enforcePrincipalBudget(owner: string, bytes: number): boolean { + while ( + principalBytes(owner) + bytes > MAX_CCR_PRINCIPAL_BYTES && + evictOldestMatching((entry) => entry.principalId === owner) + ) { + // Evict the owner's least-recently-used entries until this block fits. + } + return principalBytes(owner) + bytes <= MAX_CCR_PRINCIPAL_BYTES; +} + +function enforceGlobalBudget(bytes: number): boolean { + while ( + (ccrStore.size >= MAX_CCR_ENTRIES || ccrTotalBytes + bytes > MAX_CCR_GLOBAL_BYTES) && + evictOldestMatching(() => true) + ) { + // Enforce both entry and global byte caps with LRU eviction. + } + return ccrTotalBytes + bytes <= MAX_CCR_GLOBAL_BYTES; +} + /** * Store a block in the CCR store under the given principal. * Returns the 24-hex content hash (for embedding in the marker). */ -export function storeBlock(text: string, principalId?: string): string { +export function tryStoreBlock( + text: string, + principalId?: string, + options: StoreCcrBlockOptions = {} +): StoreCcrBlockResult { const hash = hashContent(text); + const owner = principalId ?? ANON; const key = buildStoreKey(hash, principalId); - if (!ccrStore.has(key)) { - boundedSet(ccrStore, key, text); + const now = options.now ?? Date.now(); + purgeExpired(now); + + const existing = ccrStore.get(key); + if (existing) { + existing.lastAccessedAt = now; + existing.expiresAt = now + normalizeTtlSeconds(options.ttlSeconds) * 1000; + ccrStore.delete(key); + ccrStore.set(key, existing); + return { stored: true, hash, metadata: publicMetadata(existing) }; + } + + const bytes = Buffer.byteLength(text, "utf8"); + if (bytes > MAX_CCR_BLOCK_BYTES) { + return rejectStore(hash, owner, "block_too_large"); + } + + if (!enforcePrincipalBudget(owner, bytes)) { + return rejectStore(hash, owner, "principal_budget_exceeded"); + } + + if (!enforceGlobalBudget(bytes)) { + return rejectStore(hash, owner, "global_budget_exceeded"); } - return hash; + + const ttlSeconds = normalizeTtlSeconds(options.ttlSeconds); + const entry: CcrEntry = { + hash, + principalId: owner, + content: text, + bytes, + chars: text.length, + lines: text.length === 0 ? 0 : text.split("\n").length, + contentType: options.contentType?.trim().slice(0, 128) || "text/plain", + source: options.source ?? "compression", + createdAt: now, + lastAccessedAt: now, + expiresAt: now + ttlSeconds * 1000, + }; + ccrStore.set(key, entry); + ccrTotalBytes += bytes; + principalBytesMap.set(owner, principalBytes(owner) + bytes); + return { stored: true, hash, metadata: publicMetadata(entry) }; +} + +export function storeBlock( + text: string, + principalId?: string, + options: StoreCcrBlockOptions = {} +): string { + const result = tryStoreBlock(text, principalId, options); + if (!result.stored) throw new RangeError(`CCR store rejected block: ${result.reason}`); + return result.hash; } /** * Retrieve the verbatim block for a given hash and principal. * Returns null if not found or if the principal does not match the stored key. */ -export function retrieveBlock(hash: string, principalId?: string): string | null { +export function retrieveBlock(hash: string, principalId?: string, now = Date.now()): string | null { const key = buildStoreKey(hash, principalId); - return ccrStore.get(key) ?? null; + const entry = getActiveEntry(key, now); + if (!entry) return null; + entry.lastAccessedAt = now; + ccrStore.delete(key); + ccrStore.set(key, entry); + return entry.content; } /** @@ -134,7 +352,7 @@ export function retrieveBlock(hash: string, principalId?: string): string | null */ export function recordRetrieval(hash: string, principalId?: string): void { const key = buildStoreKey(hash, principalId); - boundedSet(retrievalCounts, key, (retrievalCounts.get(key) ?? 0) + 1); + setRetrievalCount(key, (retrievalCounts.get(key) ?? 0) + 1); } /** @@ -182,6 +400,65 @@ export function resolveRetrievalRampFactor(env: NodeJS.ProcessEnv = process.env) export function resetCcrStore(): void { ccrStore.clear(); retrievalCounts.clear(); + principalBytesMap.clear(); + ccrTotalBytes = 0; + lifecycleByPrincipal.clear(); +} + +export function inspectCcrBlock( + hash: string, + principalId?: string, + now = Date.now() +): CcrEntryMetadata | null { + const entry = getActiveEntry(buildStoreKey(hash, principalId), now); + return entry ? publicMetadata(entry) : null; +} + +export function listCcrBlocks( + principalId?: string, + options: { offset?: number; limit?: number; now?: number } = {} +): { entries: CcrEntryMetadata[]; total: number; offset: number; limit: number; hasMore: boolean } { + purgeExpired(options.now); + const owner = principalId ?? ANON; + const offset = Math.max(0, Math.floor(options.offset ?? 0)); + const limit = Math.max(1, Math.min(100, Math.floor(options.limit ?? 25))); + const all: CcrEntryMetadata[] = []; + for (const entry of ccrStore.values()) { + if (entry.principalId === owner) all.push(publicMetadata(entry)); + } + all.reverse(); + return { + entries: all.slice(offset, offset + limit), + total: all.length, + offset, + limit, + hasMore: offset + limit < all.length, + }; +} + +export function deleteCcrBlock(hash: string, principalId?: string, _now = Date.now()): boolean { + return removeEntry(buildStoreKey(hash, principalId)); +} + +export function getCcrStoreStats(principalId?: string, now = Date.now()): CcrStoreStats { + purgeExpired(now); + const owner = principalId ?? ANON; + const entries = Array.from(ccrStore.values()).filter((entry) => entry.principalId === owner); + return { + storage: "memory", + entries: entries.length, + bytes: entries.reduce((sum, entry) => sum + entry.bytes, 0), + limits: { + maxEntries: MAX_CCR_ENTRIES, + maxBlockBytes: MAX_CCR_BLOCK_BYTES, + maxPrincipalBytes: MAX_CCR_PRINCIPAL_BYTES, + maxGlobalBytes: MAX_CCR_GLOBAL_BYTES, + defaultTtlSeconds: DEFAULT_CCR_TTL_SECONDS, + maxTtlSeconds: MAX_CCR_TTL_SECONDS, + maxMcpFullBytes: MAX_CCR_MCP_FULL_BYTES, + }, + lifecycle: { ...readLifecycleCounters(owner) }, + }; } // ─── MCP tool handler (pure function) ──────────────────────────────────────── @@ -226,10 +503,21 @@ type MessageLike = { /** * Build a CCR marker string for a block. */ -function buildMarker(hash: string, charCount: number): string { +export function buildCcrMarker(hash: string, charCount: number): string { return `[CCR retrieve hash=${hash} chars=${charCount}]`; } +export function buildCcrReference( + hash: string, + charCount: number +): { + hash: string; + uri: string; + marker: string; +} { + return { hash, uri: `ccr://${hash}`, marker: buildCcrMarker(hash, charCount) }; +} + /** * Replace a large text block with a CCR marker if it shrinks the content. * Returns the new text and a flag indicating whether replacement happened. @@ -254,14 +542,15 @@ function maybeCcrReplace( return { text, replaced: false, hash: null }; } - const marker = buildMarker(hash, text.length); + const marker = buildCcrMarker(hash, text.length); // Only replace if it actually shrinks if (marker.length >= text.length) { return { text, replaced: false, hash: null }; } - storeBlock(text, principalId); + const stored = tryStoreBlock(text, principalId, { source: "compression" }); + if (!stored.stored) return { text, replaced: false, hash: null }; return { text: marker, replaced: true, hash }; } diff --git a/open-sse/services/compression/engines/ionizer/sample.ts b/open-sse/services/compression/engines/ionizer/sample.ts index 172429d2a86..34fb0f38a17 100644 --- a/open-sse/services/compression/engines/ionizer/sample.ts +++ b/open-sse/services/compression/engines/ionizer/sample.ts @@ -1,5 +1,5 @@ // open-sse/services/compression/engines/ionizer/sample.ts -import { storeBlock } from "../ccr/index.ts"; +import { tryStoreBlock } from "../ccr/index.ts"; type MessageLike = { role?: string; content?: unknown; [key: string]: unknown }; @@ -131,8 +131,7 @@ export interface IonizerPassResult { function isPlainObjectArray(v: unknown): v is Array> { return ( - Array.isArray(v) && - v.every((el) => el !== null && typeof el === "object" && !Array.isArray(el)) + Array.isArray(v) && v.every((el) => el !== null && typeof el === "object" && !Array.isArray(el)) ); } @@ -169,8 +168,12 @@ export function applyIonizerPass( }); if (res.keptCount >= res.totalCount) return m; - const hash = storeBlock(serialized, opts.principalId); - const marker = `[ionizer: kept ${res.keptCount}/${res.totalCount} rows; full → CCR retrieve hash=${hash} chars=${serialized.length}]`; + const stored = tryStoreBlock(serialized, opts.principalId, { + contentType: "application/json", + source: "ionizer", + }); + if (!stored.stored) return m; + const marker = `[ionizer: kept ${res.keptCount}/${res.totalCount} rows; full → CCR retrieve hash=${stored.hash} chars=${serialized.length}]`; const newContent = `${JSON.stringify(res.kept)}\n${marker}`; if (newContent.length >= serialized.length) return m; @@ -194,7 +197,9 @@ export function runIonizerPass( principalId?: string ): IonizerPassResult { if (stepConfig["enabled"] === false) return { messages, ionizedCount: 0 }; - const threshold = typeof stepConfig["threshold"] === "number" ? (stepConfig["threshold"] as number) : 200; - const targetRows = typeof stepConfig["targetRows"] === "number" ? (stepConfig["targetRows"] as number) : 50; + const threshold = + typeof stepConfig["threshold"] === "number" ? (stepConfig["threshold"] as number) : 200; + const targetRows = + typeof stepConfig["targetRows"] === "number" ? (stepConfig["targetRows"] as number) : 50; return applyIonizerPass(messages, { threshold, targetRows, principalId }); } diff --git a/open-sse/services/compression/engines/rtk/filterLoader.ts b/open-sse/services/compression/engines/rtk/filterLoader.ts index aa10dd8224f..184d94417f0 100644 --- a/open-sse/services/compression/engines/rtk/filterLoader.ts +++ b/open-sse/services/compression/engines/rtk/filterLoader.ts @@ -4,6 +4,7 @@ import os from "node:os"; import crypto from "node:crypto"; import { detectCommandType } from "./commandDetector.ts"; import { validateRtkFilter, type RtkFilterDefinition } from "./filterSchema.ts"; +import { parseRtkTomlV1, RtkTomlCompatibilityError } from "./tomlCompatibility.ts"; let cache: RtkFilterDefinition[] | null = null; let cacheKey: string | null = null; @@ -29,6 +30,7 @@ function cachedMatchPattern(pattern: string, value: string): boolean { export interface RtkFilterLoadDiagnostic { source: "project" | "global" | "builtin"; + format?: "omniroute-json" | "rtk-toml-v1"; path?: string; level: "warning" | "error"; message: string; @@ -38,6 +40,7 @@ interface FilterSource { source: "project" | "global" | "builtin"; path: string; trusted: boolean; + format: "omniroute-json" | "rtk-toml-v1"; } interface RtkFilterLoadOptions { @@ -95,8 +98,12 @@ function projectFiltersTrusted( try { const filtersHash = sha256(fs.readFileSync(filtersPath, "utf8")); const trust = JSON.parse(fs.readFileSync(trustPath, "utf8")) as Record; - const trustedHash = - typeof trust.filtersSha256 === "string" + const isToml = filtersPath.endsWith(".toml"); + const trustedHash = isToml + ? typeof trust.filtersTomlSha256 === "string" + ? trust.filtersTomlSha256 + : null + : typeof trust.filtersSha256 === "string" ? trust.filtersSha256 : typeof trust.trustedFiltersSha256 === "string" ? trust.trustedFiltersSha256 @@ -110,29 +117,52 @@ function projectFiltersTrusted( function collectFilterSources(options: RtkFilterLoadOptions = {}): FilterSource[] { const sources: FilterSource[] = []; - const projectPath = path.join(process.cwd(), ".rtk", "filters.json"); - if (options.customFiltersEnabled !== false && fs.existsSync(projectPath)) { - const trusted = projectFiltersTrusted(projectPath, options.trustProjectFilters === true); + if (options.customFiltersEnabled !== false) { + collectProjectFilterSources(sources, options); + collectGlobalFilterSources(sources); + } + collectBuiltinFilterSources(sources); + return sources; +} + +function collectProjectFilterSources(sources: FilterSource[], options: RtkFilterLoadOptions): void { + const projectCandidates = [ + { path: path.join(process.cwd(), ".rtk", "filters.toml"), format: "rtk-toml-v1" as const }, + { path: path.join(process.cwd(), ".rtk", "filters.json"), format: "omniroute-json" as const }, + ]; + for (const candidate of projectCandidates) { + if (!fs.existsSync(candidate.path)) continue; + const trusted = projectFiltersTrusted(candidate.path, options.trustProjectFilters === true); if (trusted === true) { - sources.push({ source: "project", path: projectPath, trusted: true }); - } else { - diagnostics.push({ - source: "project", - path: projectPath, - level: "warning", - message: - trusted === "changed" - ? "Project RTK filters changed after trust and were skipped" - : "Project RTK filters are untrusted and were skipped", - }); + sources.push({ source: "project", ...candidate, trusted: true }); + continue; } + diagnostics.push({ + source: "project", + format: candidate.format, + path: candidate.path, + level: "warning", + message: + trusted === "changed" + ? "Project RTK filters changed after trust and were skipped" + : "Project RTK filters are untrusted and were skipped", + }); } +} - const globalPath = path.join(getDataDir(), "rtk", "filters.json"); - if (options.customFiltersEnabled !== false && fs.existsSync(globalPath)) { - sources.push({ source: "global", path: globalPath, trusted: true }); +function collectGlobalFilterSources(sources: FilterSource[]): void { + const globalCandidates = [ + { path: path.join(getDataDir(), "rtk", "filters.toml"), format: "rtk-toml-v1" as const }, + { path: path.join(getDataDir(), "rtk", "filters.json"), format: "omniroute-json" as const }, + ]; + for (const candidate of globalCandidates) { + if (fs.existsSync(candidate.path)) { + sources.push({ source: "global", ...candidate, trusted: true }); + } } +} +function collectBuiltinFilterSources(sources: FilterSource[]): void { const builtinDir = getFiltersDir(); if (fs.existsSync(builtinDir)) { let builtinFiles: string[] = []; @@ -152,25 +182,56 @@ function collectFilterSources(options: RtkFilterLoadOptions = {}): FilterSource[ source: "builtin", path: path.join(builtinDir, file), trusted: true, + format: "omniroute-json", }); } } - - return sources; } function parseFilterFile(source: FilterSource): RtkFilterDefinition[] { try { - const parsed = JSON.parse(fs.readFileSync(source.path, "utf8")); - const entries = Array.isArray(parsed) ? parsed : [parsed]; - return entries.map(validateRtkFilter); + const content = fs.readFileSync(source.path, "utf8"); + const definitions = + source.format === "rtk-toml-v1" + ? (() => { + const result = parseRtkTomlV1(content); + if (!result.passed) { + throw new Error("one or more inline tests failed"); + } + for (const warning of result.warnings) { + diagnostics.push({ + source: source.source, + format: source.format, + path: source.path, + level: "warning", + message: warning, + }); + } + return result.filters; + })() + : (() => { + const parsed = JSON.parse(content); + const entries = Array.isArray(parsed) ? parsed : [parsed]; + return entries.map(validateRtkFilter); + })(); + return definitions.map((definition) => ({ + ...definition, + source: source.source, + sourceFormat: definition.sourceFormat ?? source.format, + })); } catch (error) { - const message = error instanceof Error ? error.message : String(error); + const message = + error instanceof RtkTomlCompatibilityError + ? error.publicMessage + : error instanceof Error + ? error.message + : String(error); if (source.source === "builtin") { throw new Error(`Invalid RTK filter ${path.basename(source.path)}: ${message}`); } diagnostics.push({ source: source.source, + format: source.format, path: source.path, level: "warning", message: `Invalid custom RTK filter skipped: ${message}`, @@ -194,7 +255,16 @@ export function loadRtkFilters(options: RtkFilterLoadOptions = {}): RtkFilterDef filters.push(...parseFilterFile(source)); } - const sorted = filters.sort((a, b) => b.priority - a.priority || a.id.localeCompare(b.id)); + const sourceRank = { project: 3, global: 2, builtin: 1 } as const; + const formatRank = { "rtk-toml-v1": 2, "omniroute-json": 1 } as const; + const sorted = filters.sort( + (a, b) => + sourceRank[b.source ?? "builtin"] - sourceRank[a.source ?? "builtin"] || + formatRank[b.sourceFormat ?? "omniroute-json"] - + formatRank[a.sourceFormat ?? "omniroute-json"] || + b.priority - a.priority || + a.id.localeCompare(b.id) + ); cache = sorted; cacheKey = currentCacheKey; return sorted; @@ -208,7 +278,14 @@ export function getRtkFilterLoadDiagnostics(): RtkFilterLoadDiagnostic[] { export function getRtkFilterCatalog(): Array< Pick< RtkFilterDefinition, - "id" | "name" | "description" | "commandTypes" | "category" | "priority" + | "id" + | "name" + | "description" + | "commandTypes" + | "category" + | "priority" + | "source" + | "sourceFormat" > > { return loadRtkFilters().map((filter) => ({ @@ -218,6 +295,8 @@ export function getRtkFilterCatalog(): Array< commandTypes: filter.commandTypes, category: filter.category, priority: filter.priority, + source: filter.source, + sourceFormat: filter.sourceFormat, })); } @@ -229,17 +308,25 @@ export function matchRtkFilter( const detection = detectCommandType(text, command); const detectedCommand = detection.command ?? command ?? ""; const filters = loadRtkFilters(options); - return ( - filters.find((filter) => filter.commandTypes.includes(detection.type)) ?? - filters.find( - (filter) => - detectedCommand && - filter.commandPatterns.some((pattern) => cachedMatchPattern(pattern, detectedCommand)) - ) ?? - filters.find((filter) => - filter.matchPatterns.some((pattern) => cachedMatchPattern(pattern, text)) - ) ?? - filters.find((filter) => filter.commandTypes.includes("generic-output")) ?? - null - ); + for (const source of ["project", "global", "builtin"] as const) { + const scoped = filters.filter((filter) => (filter.source ?? "builtin") === source); + const matched = + scoped.find( + (filter) => + filter.sourceFormat === "rtk-toml-v1" && + detectedCommand && + filter.commandPatterns.some((pattern) => cachedMatchPattern(pattern, detectedCommand)) + ) ?? + scoped.find((filter) => filter.commandTypes.includes(detection.type)) ?? + scoped.find( + (filter) => + detectedCommand && + filter.commandPatterns.some((pattern) => cachedMatchPattern(pattern, detectedCommand)) + ) ?? + scoped.find((filter) => + filter.matchPatterns.some((pattern) => cachedMatchPattern(pattern, text)) + ); + if (matched) return matched; + } + return filters.find((filter) => filter.commandTypes.includes("generic-output")) ?? null; } diff --git a/open-sse/services/compression/engines/rtk/filterSchema.ts b/open-sse/services/compression/engines/rtk/filterSchema.ts index 2acceb7cabe..9eb9d5b621c 100644 --- a/open-sse/services/compression/engines/rtk/filterSchema.ts +++ b/open-sse/services/compression/engines/rtk/filterSchema.ts @@ -137,6 +137,12 @@ export interface RtkFilterDefinition { maxLines: number; preserveHead: number; preserveTail: number; + /** Exact RTK TOML schema-v1 head/tail stages. Undefined for OmniRoute-native JSON filters. */ + rtkTomlHeadLines?: number; + rtkTomlTailLines?: number; + rtkTomlMaxLines?: number; + sourceFormat?: "omniroute-json" | "rtk-toml-v1"; + source?: "project" | "global" | "builtin"; tests: Array<{ name: string; input: string; expected: string; command?: string }>; } diff --git a/open-sse/services/compression/engines/rtk/lineFilter.ts b/open-sse/services/compression/engines/rtk/lineFilter.ts index f1311858ba8..52a6a83870d 100644 --- a/open-sse/services/compression/engines/rtk/lineFilter.ts +++ b/open-sse/services/compression/engines/rtk/lineFilter.ts @@ -54,6 +54,41 @@ function normalizeStderrPrefix(line: string): string { return line.replace(/^\s*(?:stderr|err)\s*(?:\||:)\s*/i, ""); } +function applyRtkTomlLineLimits( + lines: string[], + filter: RtkFilterDefinition, + appliedRules: string[] +): string[] { + const head = filter.rtkTomlHeadLines; + const tail = filter.rtkTomlTailLines; + const total = lines.length; + + if (head !== undefined && tail !== undefined) { + if (total > head + tail) { + lines = [ + ...lines.slice(0, head), + `... (${total - head - tail} lines omitted)`, + ...(tail > 0 ? lines.slice(-tail) : []), + ]; + appliedRules.push(`${filter.id}:rtk-head-tail`); + } + } else if (head !== undefined && total > head) { + lines = [...lines.slice(0, head), `... (${total - head} lines omitted)`]; + appliedRules.push(`${filter.id}:rtk-head`); + } else if (tail !== undefined && total > tail) { + lines = [`... (${total - tail} lines omitted)`, ...(tail > 0 ? lines.slice(-tail) : [])]; + appliedRules.push(`${filter.id}:rtk-tail`); + } + + const maxLines = filter.rtkTomlMaxLines; + if (maxLines !== undefined && lines.length > maxLines) { + const dropped = lines.length - maxLines; + lines = [...lines.slice(0, maxLines), `... (${dropped} lines truncated)`]; + appliedRules.push(`${filter.id}:rtk-max-lines`); + } + return lines; +} + function truncateUnicodeSafe(line: string, maxChars: number): string { if (maxChars <= 0) return line; const chars = Array.from(line); @@ -70,6 +105,7 @@ export function applyLineFilter(text: string, filter: RtkFilterDefinition): Line const appliedRules: string[] = []; let lines = text.split(/\r?\n/); + if (filter.sourceFormat === "rtk-toml-v1" && lines.at(-1) === "") lines.pop(); const originalLineCount = lines.length; if (filter.stripAnsi) { @@ -159,6 +195,18 @@ export function applyLineFilter(text: string, filter: RtkFilterDefinition): Line } } + if (filter.sourceFormat === "rtk-toml-v1") { + lines = applyRtkTomlLineLimits(lines, filter, appliedRules); + const output = lines.join("\n"); + const finalOutput = output.trim().length === 0 && filter.onEmpty ? filter.onEmpty : output; + return { + text: finalOutput, + strippedLines: Math.max(0, originalLineCount - finalOutput.split(/\r?\n/).length), + keptByRule: keepPatterns.length > 0, + appliedRules, + }; + } + const truncated = smartTruncate(lines.join("\n"), { maxLines: filter.maxLines, preserveHead: filter.preserveHead, diff --git a/open-sse/services/compression/engines/rtk/tomlCompatibility.ts b/open-sse/services/compression/engines/rtk/tomlCompatibility.ts new file mode 100644 index 00000000000..1c45c045490 --- /dev/null +++ b/open-sse/services/compression/engines/rtk/tomlCompatibility.ts @@ -0,0 +1,334 @@ +import crypto from "node:crypto"; +import fs from "node:fs"; +import path from "node:path"; +import os from "node:os"; +import { parse as parseToml } from "smol-toml"; +import { z } from "zod"; +import { isReDoSProne, type RtkFilterDefinition } from "./filterSchema.ts"; +import { applyLineFilter } from "./lineFilter.ts"; + +const MAX_TOML_BYTES = 1024 * 1024; + +const replaceRuleSchema = z + .object({ + pattern: z.string().min(1), + replacement: z.string(), + }) + .strict(); + +const matchOutputRuleSchema = z + .object({ + pattern: z.string().min(1), + message: z.string(), + unless: z.string().min(1).optional(), + }) + .strict(); + +const filterSchema = z + .object({ + description: z.string().optional(), + match_command: z.string().min(1), + strip_ansi: z.boolean().optional(), + filter_stderr: z.boolean().optional(), + strip_lines_matching: z.array(z.string()).optional(), + keep_lines_matching: z.array(z.string()).optional(), + replace: z.array(replaceRuleSchema).optional(), + match_output: z.array(matchOutputRuleSchema).optional(), + truncate_lines_at: z.number().int().min(0).optional(), + head_lines: z.number().int().min(0).optional(), + tail_lines: z.number().int().min(0).optional(), + max_lines: z.number().int().min(0).optional(), + on_empty: z.string().optional(), + }) + .strict(); + +const inlineTestSchema = z + .object({ + name: z.string().min(1), + input: z.string(), + expected: z.string(), + }) + .strict(); + +const fileSchema = z + .object({ + schema_version: z.literal(1), + filters: z.record(z.string().min(1), filterSchema).default({}), + tests: z.record(z.string().min(1), z.array(inlineTestSchema)).default({}), + }) + .strict(); + +type ParsedFilter = z.infer; + +function arrayOrEmpty(value: T[] | undefined): T[] { + return value ?? []; +} + +function numberOrZero(value: number | undefined): number { + return value ?? 0; +} + +export interface RtkTomlTestOutcome { + filterId: string; + testName: string; + passed: boolean; + actual: string; + expected: string; +} + +export interface RtkTomlCompatibilityResult { + schemaVersion: 1; + sha256: string; + filters: RtkFilterDefinition[]; + outcomes: RtkTomlTestOutcome[]; + filtersWithoutTests: string[]; + warnings: string[]; + passed: boolean; +} + +export class RtkTomlCompatibilityError extends Error { + readonly publicMessage: string; + + constructor(message: string) { + super("RTK TOML schema v1 compatibility error"); + this.name = "RtkTomlCompatibilityError"; + this.publicMessage = message; + } +} + +function compatibilityError(message: string): RtkTomlCompatibilityError { + return new RtkTomlCompatibilityError(message); +} + +function categoryFor(name: string, commandPattern: string): RtkFilterDefinition["category"] { + const value = `${name} ${commandPattern}`.toLowerCase(); + if (/\b(?:git|gh)\b/.test(value)) return "git"; + if (/\b(?:test|jest|vitest|pytest|cargo test|go test|rspec|playwright)\b/.test(value)) { + return "test"; + } + if (/\b(?:build|tsc|eslint|ruff|clippy|gradle|make|next|vite|webpack)\b/.test(value)) { + return "build"; + } + if (/\b(?:docker|kubectl|podman|compose)\b/.test(value)) return "docker"; + if (/\b(?:npm|pnpm|yarn|bun|pip|poetry|uv|bundle|composer)\b/.test(value)) { + return "package"; + } + if (/\b(?:terraform|tofu|ansible|helm|pulumi)\b/.test(value)) return "infra"; + if (/\b(?:aws|gcloud|az|cloudflare)\b/.test(value)) return "cloud"; + if (/\b(?:ls|find|grep|rg|df|du|ps|systemctl|ssh|rsync)\b/.test(value)) return "shell"; + return "generic"; +} + +function regexFields(filter: ParsedFilter): Array<{ field: string; pattern: string }> { + return [ + { field: "match_command", pattern: filter.match_command }, + ...(filter.strip_lines_matching ?? []).map((pattern) => ({ + field: "strip_lines_matching", + pattern, + })), + ...(filter.keep_lines_matching ?? []).map((pattern) => ({ + field: "keep_lines_matching", + pattern, + })), + ...(filter.replace ?? []).map(({ pattern }) => ({ field: "replace.pattern", pattern })), + ...(filter.match_output ?? []).flatMap(({ pattern, unless }) => [ + { field: "match_output.pattern", pattern }, + ...(unless ? [{ field: "match_output.unless", pattern: unless }] : []), + ]), + ]; +} + +function validateRegexes(name: string, filter: ParsedFilter): void { + for (const { field, pattern } of regexFields(filter)) { + if (isReDoSProne(pattern)) { + throw compatibilityError(`filter '${name}' has an unsafe regex in ${field}`); + } + try { + new RegExp(pattern); + } catch { + throw compatibilityError(`filter '${name}' has an invalid regex in ${field}`); + } + } +} + +function toDefinition( + name: string, + filter: ParsedFilter, + tests: z.infer[] +): RtkFilterDefinition { + if ( + (filter.strip_lines_matching?.length ?? 0) > 0 && + (filter.keep_lines_matching?.length ?? 0) > 0 + ) { + throw compatibilityError( + `filter '${name}' cannot combine strip_lines_matching with keep_lines_matching` + ); + } + validateRegexes(name, filter); + return { + id: name, + name, + description: filter.description ?? "", + commandTypes: [], + commandPatterns: [filter.match_command], + matchPatterns: [], + category: categoryFor(name, filter.match_command), + priority: 50, + stripPatterns: arrayOrEmpty(filter.strip_lines_matching), + keepPatterns: arrayOrEmpty(filter.keep_lines_matching), + priorityPatterns: [], + collapsePatterns: [], + stripAnsi: filter.strip_ansi ?? false, + replace: arrayOrEmpty(filter.replace), + matchOutput: arrayOrEmpty(filter.match_output), + truncateLineAt: numberOrZero(filter.truncate_lines_at), + onEmpty: filter.on_empty ?? "", + filterStderr: false, + deduplicate: false, + maxLines: numberOrZero(filter.max_lines), + preserveHead: 0, + preserveTail: 0, + rtkTomlHeadLines: filter.head_lines, + rtkTomlTailLines: filter.tail_lines, + rtkTomlMaxLines: filter.max_lines, + sourceFormat: "rtk-toml-v1", + tests, + }; +} + +function comparable(value: string): string { + return value.replace(/\n+$/g, ""); +} + +function tomlSyntaxLocation(error: unknown): string { + if (typeof error !== "object" || error === null) return ""; + const { line, column } = error as { line?: unknown; column?: unknown }; + if (!Number.isSafeInteger(line) || !Number.isSafeInteger(column)) return ""; + return ` (line ${line}, column ${column})`; +} + +export function parseRtkTomlV1(content: string): RtkTomlCompatibilityResult { + if (Buffer.byteLength(content, "utf8") > MAX_TOML_BYTES) { + throw compatibilityError(`file exceeds the ${MAX_TOML_BYTES}-byte limit`); + } + + let raw: unknown; + try { + raw = parseToml(content); + } catch (error) { + throw compatibilityError(`invalid TOML syntax${tomlSyntaxLocation(error)}`); + } + + const parsed = fileSchema.safeParse(raw); + if (!parsed.success) { + const issue = parsed.error.issues[0]; + const field = issue?.path.length ? issue.path.join(".") : "document"; + throw compatibilityError(`${field}: ${issue?.message ?? "invalid document"}`); + } + if (Object.keys(parsed.data.filters).length === 0) { + throw compatibilityError("document contains no filters"); + } + + for (const testName of Object.keys(parsed.data.tests)) { + if (!(testName in parsed.data.filters)) { + throw compatibilityError(`tests reference unknown filter '${testName}'`); + } + } + + const filters = Object.entries(parsed.data.filters).map(([name, filter]) => + toDefinition(name, filter, parsed.data.tests[name] ?? []) + ); + const outcomes = filters.flatMap((filter) => + filter.tests.map((test) => { + const actual = comparable(applyLineFilter(test.input, filter).text); + const expected = comparable(test.expected); + return { + filterId: filter.id, + testName: test.name, + passed: actual === expected, + actual, + expected, + }; + }) + ); + const filtersWithoutTests = filters + .filter((filter) => filter.tests.length === 0) + .map((filter) => filter.id); + const warnings = filtersWithoutTests.map( + (id) => `Filter '${id}' has no inline tests and should be reviewed before installation` + ); + for (const [id, filter] of Object.entries(parsed.data.filters)) { + if (filter.filter_stderr) { + warnings.push( + `Filter '${id}': filter_stderr is accepted as a no-op because OmniRoute receives already-captured tool output` + ); + } + } + + return { + schemaVersion: 1, + sha256: crypto.createHash("sha256").update(content).digest("hex"), + filters, + outcomes, + filtersWithoutTests, + warnings, + passed: outcomes.every((outcome) => outcome.passed), + }; +} + +function getDataDir(): string { + return process.env.DATA_DIR || path.join(os.homedir(), ".omniroute"); +} + +export function getGlobalRtkTomlPath(): string { + return path.join(getDataDir(), "rtk", "filters.toml"); +} + +export function installGlobalRtkTomlV1( + content: string, + options: { overwrite?: boolean } = {} +): RtkTomlCompatibilityResult & { installedPath: string; backupCreated: boolean } { + const result = parseRtkTomlV1(content); + if (!result.passed) { + throw compatibilityError("one or more inline tests failed"); + } + + const target = getGlobalRtkTomlPath(); + const directory = path.dirname(target); + fs.mkdirSync(directory, { recursive: true, mode: 0o700 }); + try { + fs.chmodSync(directory, 0o700); + } catch { + // Best effort on filesystems that do not support POSIX permissions. + } + if (fs.existsSync(target) && !options.overwrite) { + throw compatibilityError("filters.toml already exists; confirm overwrite to replace it"); + } + + let backupCreated = false; + if (fs.existsSync(target)) { + fs.copyFileSync(target, `${target}.bak`); + try { + fs.chmodSync(`${target}.bak`, 0o600); + } catch { + // Best effort on filesystems that do not support POSIX permissions. + } + backupCreated = true; + } + const temporary = `${target}.${process.pid}.${crypto.randomUUID()}.tmp`; + try { + fs.writeFileSync(temporary, content, { encoding: "utf8", mode: 0o600, flag: "wx" }); + fs.renameSync(temporary, target); + try { + fs.chmodSync(target, 0o600); + } catch { + // Best effort on filesystems that do not support POSIX permissions. + } + } finally { + fs.rmSync(temporary, { force: true }); + } + + return { ...result, installedPath: "rtk/filters.toml", backupCreated }; +} + +export const RTK_TOML_MAX_BYTES = MAX_TOML_BYTES; diff --git a/open-sse/services/compression/engines/session-dedup/fuzzy.ts b/open-sse/services/compression/engines/session-dedup/fuzzy.ts index f61c2521c72..2f5d37f99b5 100644 --- a/open-sse/services/compression/engines/session-dedup/fuzzy.ts +++ b/open-sse/services/compression/engines/session-dedup/fuzzy.ts @@ -1,5 +1,5 @@ // open-sse/services/compression/engines/session-dedup/fuzzy.ts -import { storeBlock } from "../ccr/index.ts"; +import { buildCcrMarker, tryStoreBlock } from "../ccr/index.ts"; type MessageLike = { role?: string; content?: unknown; [key: string]: unknown }; @@ -111,8 +111,9 @@ export function applyFuzzyPass(messages: MessageLike[], opts: FuzzyPassOptions): const replacements = new Map(); for (const nd of nearDups) { - const hash = storeBlock(nd.block.text, opts.principalId); - const marker = `[CCR retrieve hash=${hash} chars=${nd.block.text.length}]`; + const stored = tryStoreBlock(nd.block.text, opts.principalId, { source: "session-dedup" }); + if (!stored.stored) continue; + const marker = buildCcrMarker(stored.hash, nd.block.text.length); if (marker.length < nd.block.text.length) replacements.set(nd.block.index, marker); } if (replacements.size === 0) return { messages, fuzzyCount: 0 }; @@ -138,9 +139,7 @@ export function runFuzzyPass( principalId?: string ): FuzzyPassResult { const raw = stepConfig["fuzzy"] as - | boolean - | { enabled?: boolean; minJaccard?: number; shingleSize?: number } - | undefined; + boolean | { enabled?: boolean; minJaccard?: number; shingleSize?: number } | undefined; const cfg = typeof raw === "boolean" ? { enabled: raw } : raw; if (!cfg?.enabled) return { messages, fuzzyCount: 0 }; return applyFuzzyPass(messages, { diff --git a/open-sse/services/compression/liveZone.ts b/open-sse/services/compression/liveZone.ts new file mode 100644 index 00000000000..d0c69cd03a2 --- /dev/null +++ b/open-sse/services/compression/liveZone.ts @@ -0,0 +1,397 @@ +import { createHash } from "node:crypto"; + +import { estimateCompressionTokens } from "./stats.ts"; +import type { CompressionResult, CompressionStats } from "./types.ts"; + +export interface LiveZoneOptions { + principalId?: string; + sessionId?: string; + variant: unknown; + ttlMinutes?: number; +} + +interface LiveZoneEntry { + rawItemDigests: string[]; + rawStableFieldsDigest: string; + transformedPrefix: unknown[]; + transformedStableFields: Record; + stats: CompressionStats | null; + lastAccess: number; + expiresAt: number; + bytes: number; +} + +interface LiveZoneContext { + field: "messages" | "input"; + key: string; + rawItems: unknown[]; + rawItemDigests: string[]; + rawStableFieldsDigest: string; + ttlMs: number; + now: number; +} + +const MAX_ENTRIES = 100; +const MAX_ENTRY_BYTES = 2 * 1024 * 1024; +const MAX_TOTAL_BYTES = 32 * 1024 * 1024; +const DEFAULT_TTL_MINUTES = 5; +const STABLE_PREFIX_FIELDS = [ + "system", + "systemInstruction", + "system_instruction", + "instructions", + "tools", + "tool_choice", +] as const; + +const entries = new Map(); +let totalBytes = 0; + +function serialize(value: unknown): string | null { + try { + const serialized = JSON.stringify(value); + return typeof serialized === "string" ? serialized : null; + } catch { + return null; + } +} + +function digest(value: unknown): string | null { + const serialized = serialize(value); + return serialized === null ? null : createHash("sha256").update(serialized).digest("hex"); +} + +function cloneItems(items: unknown[]): unknown[] | null { + try { + return structuredClone(items); + } catch { + const serialized = serialize(items); + if (serialized === null) return null; + try { + return JSON.parse(serialized) as unknown[]; + } catch { + return null; + } + } +} + +function cloneValue(value: T): T | null { + try { + return structuredClone(value); + } catch { + const serialized = serialize(value); + if (serialized === null) return null; + try { + return JSON.parse(serialized) as T; + } catch { + return null; + } + } +} + +function pickStableFields(body: Record): Record | null { + const fields: Record = {}; + for (const field of STABLE_PREFIX_FIELDS) { + if (Object.prototype.hasOwnProperty.call(body, field)) fields[field] = body[field]; + } + return cloneValue(fields); +} + +function sequenceField(body: Record): "messages" | "input" | null { + if (Array.isArray(body.messages)) return "messages"; + if (Array.isArray(body.input)) return "input"; + return null; +} + +function isToolOutputItem(value: unknown): boolean { + if (!value || typeof value !== "object") return false; + const item = value as Record; + return ( + item.role === "tool" || + item.role === "function" || + item.role === "tool_result" || + item.type === "function_call_output" || + item.type === "computer_call_output" || + item.type === "tool_result" + ); +} + +function makeKey(options: LiveZoneOptions, field: string): string | null { + const principal = options.principalId?.trim(); + const session = options.sessionId?.trim(); + const variant = digest(options.variant); + if (!principal || !session || !variant) return null; + return `${principal}:${session}:${field}:${variant}`; +} + +function deleteEntry(key: string): void { + const existing = entries.get(key); + if (!existing) return; + totalBytes -= existing.bytes; + entries.delete(key); +} + +function prune(now: number): void { + for (const [key, entry] of entries) { + if (now >= entry.expiresAt) deleteEntry(key); + } + while (entries.size > MAX_ENTRIES || totalBytes > MAX_TOTAL_BYTES) { + const oldest = entries.keys().next().value as string | undefined; + if (!oldest) break; + deleteEntry(oldest); + } +} + +function store( + key: string, + rawItemDigests: string[], + rawStableFieldsDigest: string, + result: CompressionResult, + field: "messages" | "input", + now: number, + ttlMs: number +): void { + const transformedItems = result.body[field]; + if (!Array.isArray(transformedItems)) return; + const transformedPrefix = cloneItems(transformedItems); + const transformedStableFields = pickStableFields(result.body); + const stats = cloneValue(result.stats); + if (!transformedPrefix || !transformedStableFields) return; + const serialized = serialize({ transformedPrefix, transformedStableFields, stats }); + if (serialized === null) return; + const bytes = Buffer.byteLength(serialized, "utf8") + rawItemDigests.length * 64; + if (bytes > MAX_ENTRY_BYTES) return; + + deleteEntry(key); + entries.set(key, { + rawItemDigests, + rawStableFieldsDigest, + transformedPrefix, + transformedStableFields, + stats, + lastAccess: now, + expiresAt: now + ttlMs, + bytes, + }); + totalBytes += bytes; + prune(now); +} + +function hasExactRawPrefix(rawItemDigests: string[], entry: LiveZoneEntry): boolean { + if (rawItemDigests.length < entry.rawItemDigests.length) return false; + for (let index = 0; index < entry.rawItemDigests.length; index++) { + if (rawItemDigests[index] !== entry.rawItemDigests[index]) return false; + } + return true; +} + +function restoreStableFields( + body: Record, + stableFields: Record +): Record | null { + const restored = cloneValue(stableFields); + return restored ? { ...body, ...restored } : null; +} + +function withLiveZoneStats( + body: Record, + result: CompressionResult, + frozenItems: number, + liveItems: number +): CompressionResult { + const originalTokens = estimateCompressionTokens(body); + const compressedTokens = estimateCompressionTokens(result.body); + const savingsPercent = + originalTokens > 0 + ? Math.max( + 0, + Math.round(((originalTokens - compressedTokens) / originalTokens) * 10000) / 100 + ) + : 0; + const base = result.stats; + const stats: CompressionStats = { + ...(base ?? { + techniquesUsed: [], + mode: "stacked", + timestamp: Date.now(), + }), + originalTokens, + compressedTokens, + savingsPercent, + techniquesUsed: [...new Set([...(base?.techniquesUsed ?? []), "live-zone-prefix-reuse"])], + liveZone: { + cacheHit: true, + frozenItems, + liveItems, + }, + }; + return { + ...result, + compressed: result.compressed || compressedTokens < originalTokens, + stats, + }; +} + +function hasGlobalHardBudget(variant: unknown): boolean { + if (!variant || typeof variant !== "object") return false; + const config = (variant as Record).config; + if (!config || typeof config !== "object") return false; + const record = config as Record; + return record.targetTokens != null || record.targetRatio != null; +} + +function resolveLiveZoneContext( + body: Record, + options: LiveZoneOptions +): LiveZoneContext | null { + const field = sequenceField(body); + const key = field ? makeKey(options, field) : null; + if (!field || !key) return null; + + const rawItems = body[field] as unknown[]; + const rawItemDigests = rawItems.map(digest); + if (rawItemDigests.some((value) => value === null)) return null; + const rawStableFieldsDigest = digest(pickStableFields(body)); + if (!rawStableFieldsDigest) return null; + const ttlMinutes = Math.min(60, Math.max(1, options.ttlMinutes ?? DEFAULT_TTL_MINUTES)); + const now = Date.now(); + return { + field, + key, + rawItems, + rawItemDigests: rawItemDigests as string[], + rawStableFieldsDigest, + ttlMs: ttlMinutes * 60_000, + now, + }; +} + +async function compressAndStore( + body: Record, + context: LiveZoneContext, + compress: (body: Record) => Promise +): Promise { + const result = await compress(body); + store( + context.key, + context.rawItemDigests, + context.rawStableFieldsDigest, + result, + context.field, + context.now, + context.ttlMs + ); + return result; +} + +async function compressLiveToolOutputs( + body: Record, + field: "messages" | "input", + liveItems: unknown[], + previousStats: CompressionStats | null, + compress: (body: Record) => Promise +): Promise<{ liveResult: CompressionResult; transformedLive: unknown[] } | null> { + const transformedLive = cloneItems(liveItems); + if (!transformedLive) return null; + const liveToolIndexes = liveItems.flatMap((item, index) => + isToolOutputItem(item) ? [index] : [] + ); + if (liveToolIndexes.length === 0) { + return { liveResult: { body, compressed: false, stats: previousStats }, transformedLive }; + } + + const liveToolItems = liveToolIndexes.map((index) => liveItems[index]); + const liveResult = await compress({ ...body, [field]: liveToolItems }); + const transformed = liveResult.body[field]; + if (!Array.isArray(transformed) || transformed.length !== liveToolItems.length) { + return { liveResult: { body, compressed: false, stats: null }, transformedLive }; + } + for (let index = 0; index < liveToolIndexes.length; index++) { + transformedLive[liveToolIndexes[index]] = transformed[index]; + } + return { liveResult, transformedLive }; +} + +async function reuseLiveZoneEntry( + body: Record, + context: LiveZoneContext, + previous: LiveZoneEntry, + compress: (body: Record) => Promise +): Promise { + entries.delete(context.key); + previous.lastAccess = context.now; + entries.set(context.key, previous); + + const frozenItems = previous.rawItemDigests.length; + const liveItems = context.rawItems.slice(frozenItems); + const frozenPrefix = cloneItems(previous.transformedPrefix); + if (!frozenPrefix) return compress(body); + const live = await compressLiveToolOutputs( + body, + context.field, + liveItems, + previous.stats, + compress + ); + if (!live) return compress(body); + const restoredBody = restoreStableFields(live.liveResult.body, previous.transformedStableFields); + if (!restoredBody) return compress(body); + const combinedBody = { + ...restoredBody, + [context.field]: [...frozenPrefix, ...live.transformedLive], + }; + const combinedResult = withLiveZoneStats( + body, + { ...live.liveResult, body: combinedBody }, + frozenItems, + liveItems.length + ); + if (entries.get(context.key) === previous) { + store( + context.key, + context.rawItemDigests, + context.rawStableFieldsDigest, + combinedResult, + context.field, + Date.now(), + context.ttlMs + ); + } + return combinedResult; +} + +/** + * Reuses the byte-identical transformed prefix from the previous request in a session and runs + * compression only over newly appended messages/input items. Any changed prefix, missing identity, + * unsupported body shape, serialization failure, or oversized entry fails open to full compression. + */ +export async function applyLiveZoneCompression( + body: Record, + options: LiveZoneOptions, + compress: (body: Record) => Promise +): Promise { + // A global hard budget needs the complete history to make correct keep/drop decisions. + if (hasGlobalHardBudget(options.variant)) return compress(body); + const context = resolveLiveZoneContext(body, options); + if (!context) return compress(body); + prune(context.now); + const previous = entries.get(context.key); + + if ( + !previous || + previous.rawStableFieldsDigest !== context.rawStableFieldsDigest || + !hasExactRawPrefix(context.rawItemDigests, previous) + ) { + return compressAndStore(body, context, compress); + } + return reuseLiveZoneEntry(body, context, previous, compress); +} + +export function resetLiveZoneCache(): void { + entries.clear(); + totalBytes = 0; +} + +export function getLiveZoneCacheStats(): { entries: number; bytes: number } { + return { entries: entries.size, bytes: totalBytes }; +} diff --git a/open-sse/services/compression/types.ts b/open-sse/services/compression/types.ts index 9bba7973b27..ff8aedd77b5 100644 --- a/open-sse/services/compression/types.ts +++ b/open-sse/services/compression/types.ts @@ -131,6 +131,11 @@ export interface ContextEditingConfig { enabled: boolean; } +/** Cache-aligned compression: freeze a previously transformed prefix and process only new items. */ +export interface LiveZoneConfig { + enabled: boolean; +} + export interface CompressionPipelineStep { engine: CompressionEngineId; intensity?: CavemanIntensity | RtkIntensity; @@ -193,6 +198,8 @@ export interface CompressionConfig { ultra?: UltraConfig; /** Provider-delegated context editing (Claude/Anthropic only). */ contextEditing?: ContextEditingConfig; + /** Opt-in cache-aligned live-zone compression (default disabled). */ + liveZone?: LiveZoneConfig; /** Per-engine opt-in toggles for the config panel. */ engines: Record; /** Active combo preset id, or null if none selected. */ @@ -294,6 +301,11 @@ export interface CompressionStats { }>; /** Present only when QuantumLock stabilized ≥1 fragment this run. */ quantumLock?: QuantumLockStats; + liveZone?: { + cacheHit: boolean; + frozenItems: number; + liveItems: number; + }; } export interface CompressionResult { @@ -321,6 +333,7 @@ export const DEFAULT_COMPRESSION_CONFIG: CompressionConfig = { activeComboId: null, ultraEngine: "heuristic", ultraSlmPrewarm: false, + liveZone: { enabled: false }, }; export const DEFAULT_CAVEMAN_CONFIG: CavemanConfig = { diff --git a/open-sse/services/responsesInputSanitizer.ts b/open-sse/services/responsesInputSanitizer.ts index c1f938a38b3..3546a294628 100644 --- a/open-sse/services/responsesInputSanitizer.ts +++ b/open-sse/services/responsesInputSanitizer.ts @@ -92,6 +92,29 @@ function sanitizeMessageContent(record: JsonRecord): JsonRecord { return { ...record, content }; } +function sanitizeNestedOutputPart(part: unknown): unknown { + const record = toRecord(part); + if (!record) return part; + + // `output` on replayed items is an input-side container. Its content uses + // input content-part types even when the enclosing item originated from an + // assistant/tool response. Converting an image placeholder to output_text + // here makes Codex reject the request with the inverse 400. + if (record.type === "output_text" || record.type === "refusal") { + const next: JsonRecord = { ...record, type: "input_text" }; + if (typeof next.text !== "string") { + next.text = typeof record.refusal === "string" ? record.refusal : ""; + } + delete next.annotations; + delete next.logprobs; + delete next.obfuscation; + delete next.refusal; + return next; + } + + return sanitizeContentPart(part, "user"); +} + function sanitizeOutputContent(record: JsonRecord): JsonRecord { if (!Array.isArray(record.output)) return record; @@ -99,8 +122,7 @@ function sanitizeOutputContent(record: JsonRecord): JsonRecord { // Responses input. In that shape OpenAI validates `input[n].output[m].type` // against output content part types, so legacy Chat-style `image_url` parts // must be normalized here too, not only in message.content. - const role = record.type === "function_call_output" ? "user" : "assistant"; - const output = record.output.map((part) => sanitizeContentPart(part, role)); + const output = record.output.map(sanitizeNestedOutputPart); return { ...record, output }; } diff --git a/open-sse/translator/request/claude-to-gemini.ts b/open-sse/translator/request/claude-to-gemini.ts index f7af11ad302..b3d987055de 100644 --- a/open-sse/translator/request/claude-to-gemini.ts +++ b/open-sse/translator/request/claude-to-gemini.ts @@ -188,7 +188,9 @@ export function claudeToGeminiRequest(model, body, stream, credentials = null) { // Priority: thinking.budget_tokens (Claude native) > output_config.effort (Claude Code). if (model.startsWith("gemma-4")) { // gemma-4 models returns - 400: Thinking budget is not supported for this model - } else if (body.thinking?.type === "enabled" && body.thinking.budget_tokens) { + } else if (body.thinking?.type === "enabled" && body.thinking.budget_tokens !== undefined) { + // #6813: a truthy check here dropped `budget_tokens: 0` (dynamic thinking). + // `undefined` (no budget specified) still falls through to the effort branch. result.generationConfig.thinkingConfig = { thinkingBudget: body.thinking.budget_tokens, includeThoughts: true, diff --git a/open-sse/translator/request/openai-responses/toResponses.ts b/open-sse/translator/request/openai-responses/toResponses.ts index bda9113d08d..48eaea8533b 100644 --- a/open-sse/translator/request/openai-responses/toResponses.ts +++ b/open-sse/translator/request/openai-responses/toResponses.ts @@ -49,6 +49,32 @@ function mapChatResponseFormatToResponsesText(body: JsonRecord, result: JsonReco result.text = { ...existingText, format }; } +// Convert a Chat-Completions content block (string or text-part array) into the +// Responses API `input_text` part array used by message input items. +function buildResponsesTextParts(content: unknown): unknown[] { + if (typeof content === "string") { + return [{ type: "input_text", text: content }]; + } + if (Array.isArray(content)) { + const parts: unknown[] = []; + for (const partValue of content) { + // A bare string inside the content array is a real text instruction + // (e.g. a harness-injected system reminder), not a structured part. + // Silently dropping it lost the instruction (#6954 follow-up). + if (typeof partValue === "string") { + parts.push({ type: "input_text", text: partValue }); + continue; + } + const part = toRecord(partValue); + if (part.type === "text" || typeof part.text === "string") { + parts.push({ type: "input_text", text: toString(part.text) }); + } + } + return parts.length > 0 ? parts : [{ type: "input_text", text: "" }]; + } + return [{ type: "input_text", text: "" }]; +} + export function openaiToOpenAIResponsesRequest( model: unknown, body: unknown, @@ -83,7 +109,18 @@ export function openaiToOpenAIResponsesRequest( if (!hasSystemMessage) { result.instructions = typeof msg.content === "string" ? msg.content : ""; hasSystemMessage = true; + continue; } + // Mid-conversation system/developer turns (e.g. harness-injected reminders + // from Claude Code) must survive as developer-role input items. The + // Responses API supports the `developer` role for exactly this; mapping + // them to `assistant` misattributes harness instructions as model output, + // and silently dropping them loses them entirely (#6954). + input.push({ + type: "message", + role: "developer", + content: buildResponsesTextParts(msg.content), + }); continue; } diff --git a/open-sse/utils/diagnostics.ts b/open-sse/utils/diagnostics.ts index 176d0f7843f..39161e41373 100644 --- a/open-sse/utils/diagnostics.ts +++ b/open-sse/utils/diagnostics.ts @@ -284,5 +284,27 @@ export function detectMalformedNonStream(resp: unknown): MalformedReason | null return null; } +export function describeMalformedNonStream( + resp: unknown, + reason: MalformedReason +): { message: string; code: string; type: string } { + const body = resp && typeof resp === "object" ? (resp as Record) : null; + if (body?.object === "response" && body.status === "failed") { + return { + message: "upstream reported a failed response without usable output", + code: "upstream_response_failed", + type: "upstream_response_error", + }; + } + return { + message: + reason === "no_terminal" + ? "upstream response did not reach a terminal state" + : "upstream returned an empty response without usable output", + code: "upstream_empty_response", + type: "upstream_response_error", + }; +} + // ── Test-only export ───────────────────────────────────────────────────────── export const __test = { describeReason }; diff --git a/open-sse/utils/sseHeartbeat.ts b/open-sse/utils/sseHeartbeat.ts index 9a122141457..005a19b1728 100644 --- a/open-sse/utils/sseHeartbeat.ts +++ b/open-sse/utils/sseHeartbeat.ts @@ -61,6 +61,20 @@ type SseHeartbeatTransformOptions = { const HEARTBEAT_ENCODER = new TextEncoder(); +/** + * Whether OmniRoute may emit SSE `:` comment lines (e.g. the `: keepalive` heartbeat). + * Some strict OpenAI-compatible clients parse every SSE line as JSON and crash on `:` comments. + * Set OMNIROUTE_SSE_COMMENTS=off to suppress comment-shaped heartbeats (they become a no-op). + * Defaults to enabled for backward compatibility. + */ +export function sseCommentsEnabled(): boolean { + // SSR/edge safety: `process` is not defined in Workers/Deno/edge runtimes. + if (typeof process === "undefined") return true; + const v = process.env.OMNIROUTE_SSE_COMMENTS; + if (v === undefined || v === "") return true; + return v.trim().toLowerCase() !== "off"; +} + export function createSseHeartbeatTransform({ intervalMs = DEFAULT_SSE_HEARTBEAT_INTERVAL_MS, signal, @@ -72,6 +86,13 @@ export function createSseHeartbeatTransform({ return new TransformStream(); } + // Opt-out for strict OpenAI-compatible clients that JSON.parse every SSE line and + // crash on `:` comment heartbeats. OMNIROUTE_SSE_COMMENTS=off disables comment-shaped + // heartbeats (they become a no-op); valid `data:` heartbeats are unaffected. + if (!sseCommentsEnabled() && shape === HEARTBEAT_SHAPES.COMMENT) { + return new TransformStream(); + } + let intervalId: ReturnType | undefined; const stop = () => { diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index 974d532e9eb..e8356db21bb 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -640,6 +640,24 @@ export function createSSEStream(options: StreamOptions = {}) { dropResponsesCommentary, } = options; const signatureNamespace = connectionId; + // Request-body-size metric (for monitoring payload size distribution & correlation with TTFT). + // The size is JSON-serialised byte count; stored as a performance mark detail so monitoring + // tools can query performance.getEntriesByType("mark") filtered by name. + let bodySize = 0; + try { + bodySize = body ? Buffer.byteLength(JSON.stringify(body), "utf8") : 0; + } catch { + /* body may not be JSON-serialisable (e.g. FormData, Blob) — metric stays 0 */ + } + if (bodySize > 0) { + // Cleared immediately: this is a fixed-name mark created on every stream, so + // leaving it in the global performance timeline would accumulate without bound + // over a long-running server's lifetime. A wired PerformanceObserver still + // receives the entry (delivery is queued independently of the buffer) even though + // clearMarks() removes it from getEntriesByName()/getEntriesByType() right after. + performance.mark("omni-request-body-size", { detail: bodySize }); + performance.clearMarks("omni-request-body-size"); + } // Drop internal commentary-phase Responses output before forwarding (#6199). // Explicit option wins; otherwise read the feature flag (default on). Resolved diff --git a/package-lock.json b/package-lock.json index a678e41868f..d5eb7bd8fd2 100644 --- a/package-lock.json +++ b/package-lock.json @@ -74,6 +74,7 @@ "recharts": "^3.8.1", "safe-regex": "^2.1.1", "selfsigned": "^5.5.0", + "smol-toml": "1.6.1", "socks": "^2.8.7", "sql.js": "^1.14.1", "sqlite-vec": "^0.1.9", @@ -103,7 +104,7 @@ "@testing-library/jest-dom": "^6.9.1", "@testing-library/react": "^16.3.2", "@types/better-sqlite3": "^7.6.13", - "@types/bun": "latest", + "@types/bun": "*", "@types/node": "^26.1.0", "@types/react": "^19.2.15", "@types/react-dom": "^19.2.3", @@ -144,7 +145,7 @@ "wtfnode": "^0.10.1" }, "engines": { - "node": ">=22.0.0 <23 || >=24.0.0 <27" + "node": ">=22.22.2 <23 || >=24.0.0 <27" }, "optionalDependencies": { "@atjsh/llmlingua-2": "2.0.3", @@ -213,17 +214,6 @@ "zod": "^3.25.76 || ^4.1.8" } }, - "node_modules/@ai-zen/node-fetch-event-source": { - "version": "2.1.4", - "resolved": "https://registry.npmjs.org/@ai-zen/node-fetch-event-source/-/node-fetch-event-source-2.1.4.tgz", - "integrity": "sha512-OHFwPJecr+qwlyX5CGmTvKAKPZAdZaxvx/XDqS1lx4I2ZAk9riU0XnEaRGOOAEFrdcLZ98O5yWqubwjaQc0umg==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "cross-fetch": "^4.0.0" - } - }, "node_modules/@alcalzone/ansi-tokenize": { "version": "0.3.0", "resolved": "https://registry.npmjs.org/@alcalzone/ansi-tokenize/-/ansi-tokenize-0.3.0.tgz", @@ -295,9 +285,9 @@ } }, "node_modules/@anthropic-ai/claude-agent-sdk": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk/-/claude-agent-sdk-0.3.195.tgz", - "integrity": "sha512-FVmXu9pvOMbuBKWrF8YsYQdQ/upOpv5rS8lFAnFO5jbyXT/2hN7kEPd2vd2GJpaMvNcO/KptyQUK5AxjjTz3+w==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk/-/claude-agent-sdk-0.3.201.tgz", + "integrity": "sha512-InT1XLmf2QpldWdtznKDWEoGJT4p+sXh24yxbeBQ++lMJCzMrI0W27MEmmmDWx0otpa+ubdHCF5YQ6oiNt7cmg==", "dev": true, "license": "SEE LICENSE IN README.md", "optional": true, @@ -305,14 +295,14 @@ "node": ">=18.0.0" }, "optionalDependencies": { - "@anthropic-ai/claude-agent-sdk-darwin-arm64": "0.3.195", - "@anthropic-ai/claude-agent-sdk-darwin-x64": "0.3.195", - "@anthropic-ai/claude-agent-sdk-linux-arm64": "0.3.195", - "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": "0.3.195", - "@anthropic-ai/claude-agent-sdk-linux-x64": "0.3.195", - "@anthropic-ai/claude-agent-sdk-linux-x64-musl": "0.3.195", - "@anthropic-ai/claude-agent-sdk-win32-arm64": "0.3.195", - "@anthropic-ai/claude-agent-sdk-win32-x64": "0.3.195" + "@anthropic-ai/claude-agent-sdk-darwin-arm64": "0.3.201", + "@anthropic-ai/claude-agent-sdk-darwin-x64": "0.3.201", + "@anthropic-ai/claude-agent-sdk-linux-arm64": "0.3.201", + "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": "0.3.201", + "@anthropic-ai/claude-agent-sdk-linux-x64": "0.3.201", + "@anthropic-ai/claude-agent-sdk-linux-x64-musl": "0.3.201", + "@anthropic-ai/claude-agent-sdk-win32-arm64": "0.3.201", + "@anthropic-ai/claude-agent-sdk-win32-x64": "0.3.201" }, "peerDependencies": { "@anthropic-ai/sdk": ">=0.93.0", @@ -321,9 +311,9 @@ } }, "node_modules/@anthropic-ai/claude-agent-sdk-darwin-arm64": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-darwin-arm64/-/claude-agent-sdk-darwin-arm64-0.3.195.tgz", - "integrity": "sha512-WIMM/8HRCLsTDHFTIwQvvE8WCA/oaMJtdQxsP7iNyfzIGwXbuOyU95V8vYIhZfaO2yaSpbBRncunq4CtR5H4ng==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-darwin-arm64/-/claude-agent-sdk-darwin-arm64-0.3.201.tgz", + "integrity": "sha512-8Mcb3BDyKUGfJWFFTWwt+at37lbDH3ZwVtUNPWGG1toZ75RDCJry5U4kXRvQ2xokvJQlA0E+eNp6keWe5ZH22Q==", "cpu": [ "arm64" ], @@ -335,9 +325,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-darwin-x64": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-darwin-x64/-/claude-agent-sdk-darwin-x64-0.3.195.tgz", - "integrity": "sha512-RY7DB+4LXosE0MJ+XELmakfPrDN1YX4lkk9CTDm28jGCVcESRz9kAEqbyaiC48dZcmN9V1NCLutzINGdcr1TBg==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-darwin-x64/-/claude-agent-sdk-darwin-x64-0.3.201.tgz", + "integrity": "sha512-TFR2bu0+ml3RHoMrtsgD0qDK5Oknw8kYGBV7qpQHn+IWmE96gnHhogG1LpJwpHtni08XkJIjfWk1DdlsUYtRkQ==", "cpu": [ "x64" ], @@ -349,9 +339,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-linux-arm64": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-arm64/-/claude-agent-sdk-linux-arm64-0.3.195.tgz", - "integrity": "sha512-JuIq5Fnz/F1snl0aqi1gcuRZqPWoPNrL9dJ0DuievCxKkO8hnEz/Mmn5Zos7x1X8HE//ZnEvmQXoEQEZXonJew==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-arm64/-/claude-agent-sdk-linux-arm64-0.3.201.tgz", + "integrity": "sha512-mShTo3MwF0gkN4dDw78wWJiB6aBDVRkl81cnApvoBofpdyUBYgm9Gw16CCjDTgelMKeBFqN6ErJpwjI3wbP00A==", "cpu": [ "arm64" ], @@ -366,9 +356,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-linux-arm64-musl": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-arm64-musl/-/claude-agent-sdk-linux-arm64-musl-0.3.195.tgz", - "integrity": "sha512-ZmyBA/AFzhgutcxb7dbhCm6GTjJytwNYXTxJoKE2B3A409WCYccjMqeji6vCMNxyyfylglGo5D8dVMIxW9aoug==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-arm64-musl/-/claude-agent-sdk-linux-arm64-musl-0.3.201.tgz", + "integrity": "sha512-EiqbpfJIpChfkn+8Uj061Qjyw0eaRcOXtdrvVuHANyj8ZErVOr8HlH6op9PSeIUa9TX0m2+tNgKPQvOGseQckA==", "cpu": [ "arm64" ], @@ -383,9 +373,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-linux-x64": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-x64/-/claude-agent-sdk-linux-x64-0.3.195.tgz", - "integrity": "sha512-s1lNi1cL93luoqsItH+fNO4KpIhdkvnVhWGGQUQ/8ftwa2gfmcIQnOg1hG8Ks+KzeD3UUQ8L9YEVHVADnFI/9A==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-x64/-/claude-agent-sdk-linux-x64-0.3.201.tgz", + "integrity": "sha512-jrJBrRWrSuoFKIgjyqxHqmfd6Pb3Bs5Bvakg0knXCTC4fbUXGnC9Q6u7gdDwgXohUNP6/DD+s8U7bivvvVv0dg==", "cpu": [ "x64" ], @@ -400,9 +390,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-linux-x64-musl": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-x64-musl/-/claude-agent-sdk-linux-x64-musl-0.3.195.tgz", - "integrity": "sha512-nf8Q/LauB+ZOC6QDjxNhbsvwUtYjKYnaWJLTYFwhkmsLujePnety1AtT/1ubaUoq5AM1j297DhMlYTasa79OUA==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-linux-x64-musl/-/claude-agent-sdk-linux-x64-musl-0.3.201.tgz", + "integrity": "sha512-IbxnzO5UCbqbm2TnzCHkSyJorAFw2isdKdIsFCTxJJjSs3ZC+v3LC1QSUiVCx0qi+CV6w3MKx6mLI11mrvhbbQ==", "cpu": [ "x64" ], @@ -417,9 +407,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-win32-arm64": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-win32-arm64/-/claude-agent-sdk-win32-arm64-0.3.195.tgz", - "integrity": "sha512-hbkDE+xPIZzRWm+D+BKrH9uJH6USIZdDIlsyrIlGi3JFHoieYoA1vdUNyldSS9+F3ZqQtfPjr2Qy08IVB6akYA==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-win32-arm64/-/claude-agent-sdk-win32-arm64-0.3.201.tgz", + "integrity": "sha512-UsoytRJ/037uHpb3ATrIoe+AgwTf+PwKuFLGjddHAV/11wERJs0hlrnSmcnp43kf0PFxoSNinngme96YYASmQg==", "cpu": [ "arm64" ], @@ -431,9 +421,9 @@ ] }, "node_modules/@anthropic-ai/claude-agent-sdk-win32-x64": { - "version": "0.3.195", - "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-win32-x64/-/claude-agent-sdk-win32-x64-0.3.195.tgz", - "integrity": "sha512-av0piEB3X1Dzhpr8A+DqHVZ9y8s1jpn8enzwX0TKKUPBn5IqLTWC7wD6v66aoUgu4f+g4ThZirmDZA6shyPEZQ==", + "version": "0.3.201", + "resolved": "https://registry.npmjs.org/@anthropic-ai/claude-agent-sdk-win32-x64/-/claude-agent-sdk-win32-x64-0.3.201.tgz", + "integrity": "sha512-PhalN/0cWcqDfbx7iwoLNR2gurjTiqhBk1G6K+NRScxEcQjWuu5xKXCcdbX8ePVpT+nbEMmFEFpn2y+8V8hIdA==", "cpu": [ "x64" ], @@ -445,9 +435,9 @@ ] }, "node_modules/@anthropic-ai/sdk": { - "version": "0.106.0", - "resolved": "https://registry.npmjs.org/@anthropic-ai/sdk/-/sdk-0.106.0.tgz", - "integrity": "sha512-ufwVvYNDBj2dzOGupBCTaNzBLxqcTnGOzI4z8Wouxlt+mT3J3HuOmatgCy1VmwCHOUueqZ41ERhm0O99OUcbWA==", + "version": "0.110.0", + "resolved": "https://registry.npmjs.org/@anthropic-ai/sdk/-/sdk-0.110.0.tgz", + "integrity": "sha512-hOP4bNYXDFHDxxiEgzlILXrxZIYCDnhe8sry0RDRKD/QnsEpvZcQpablCdm9X/WuD/YgOiSIkkqsL1mLLlTqJw==", "dev": true, "license": "MIT", "dependencies": { @@ -611,22 +601,22 @@ } }, "node_modules/@aws-sdk/client-bedrock-runtime": { - "version": "3.1081.0", - "resolved": "https://registry.npmjs.org/@aws-sdk/client-bedrock-runtime/-/client-bedrock-runtime-3.1081.0.tgz", - "integrity": "sha512-rRAGXY5qV/NCYbVA0QZHPierv3diOOiE4+1f5vedpbyvg7Phh9m/I2pRFwu0koUtMdRSGPWgoNcvMIyaDhDj7Q==", + "version": "3.1088.0", + "resolved": "https://registry.npmjs.org/@aws-sdk/client-bedrock-runtime/-/client-bedrock-runtime-3.1088.0.tgz", + "integrity": "sha512-m76gdG4tYCStbh1MjWeO8w40la/Sgs+0kpO2iWaWF09vGpZ9iwwcsR//k5g7OxIfuOWdiwP4TvgoTNaEAxy0fg==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.974.29", - "@aws-sdk/credential-provider-node": "^3.972.64", - "@aws-sdk/eventstream-handler-node": "^3.972.25", - "@aws-sdk/middleware-eventstream": "^3.972.21", - "@aws-sdk/middleware-websocket": "^3.972.37", - "@aws-sdk/token-providers": "3.1081.0", - "@aws-sdk/types": "^3.973.15", - "@smithy/core": "^3.29.0", - "@smithy/fetch-http-handler": "^5.6.2", - "@smithy/node-http-handler": "^4.9.2", - "@smithy/types": "^4.15.1", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/credential-provider-node": "^3.972.69", + "@aws-sdk/eventstream-handler-node": "^3.972.28", + "@aws-sdk/middleware-eventstream": "^3.972.24", + "@aws-sdk/middleware-websocket": "^3.972.41", + "@aws-sdk/token-providers": "3.1088.0", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/fetch-http-handler": "^5.6.6", + "@smithy/node-http-handler": "^4.9.6", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -679,17 +669,17 @@ } }, "node_modules/@aws-sdk/core": { - "version": "3.975.1", - "resolved": "https://registry.npmjs.org/@aws-sdk/core/-/core-3.975.1.tgz", - "integrity": "sha512-8qh/6EYb7hl/ZwVfQufhbMEZs1gQIc7GbdrIf4eprQJ7cv042+74nE6l3YDfyWNzb9iPXb8fRyYSHkNIk5eE6Q==", + "version": "3.975.3", + "resolved": "https://registry.npmjs.org/@aws-sdk/core/-/core-3.975.3.tgz", + "integrity": "sha512-7ur3kCKuvPLqlsZ2XlvnNBVQ7KkpSu6Y6dOTwSPHLrFpTEfZM8isLBJc4cgv96WB7GifeVM436mpycwxBd2vEA==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/types": "^3.974.0", - "@aws-sdk/xml-builder": "^3.972.34", + "@aws-sdk/types": "^3.974.2", + "@aws-sdk/xml-builder": "^3.972.36", "@aws/lambda-invoke-store": "^0.3.0", - "@smithy/core": "^3.29.2", - "@smithy/signature-v4": "^5.6.3", - "@smithy/types": "^4.16.0", + "@smithy/core": "^3.29.4", + "@smithy/signature-v4": "^5.6.5", + "@smithy/types": "^4.16.1", "bowser": "^2.11.0", "tslib": "^2.6.2" }, @@ -698,15 +688,15 @@ } }, "node_modules/@aws-sdk/credential-provider-env": { - "version": "3.972.57", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-env/-/credential-provider-env-3.972.57.tgz", - "integrity": "sha512-1RfJaF7SW1TOnvNGU7kaYjwUf5H3sfm+synGH1bHhRlqcnxCt3szebH3dmKEyY4tuGcbQ6ffzUT89cRitBV8OQ==", + "version": "3.972.59", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-env/-/credential-provider-env-3.972.59.tgz", + "integrity": "sha512-Ny5e4Mfh3QPmiAc0AiUe+cbTXDlxkU3Rc+EpWOfyWeWEy6yp7Fa1KmfNeCc+1a8by9zQ9gtohmiQUkMPScF3ng==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -714,17 +704,17 @@ } }, "node_modules/@aws-sdk/credential-provider-http": { - "version": "3.972.59", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-http/-/credential-provider-http-3.972.59.tgz", - "integrity": "sha512-sRCkpTiFnCdQvuaRVjQ6SVoHu6i7RUpurVo1c4F81HWhPvUJ7Wdp5MNtSdX1O29CNXc8em3O5m52hCjVtAD9SA==", + "version": "3.972.61", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-http/-/credential-provider-http-3.972.61.tgz", + "integrity": "sha512-8jAjgStl5Ytq4+HF3X/9f+EmRinaRbGRRtQGktlPfBRVx73H+R1y48vIeXerQtYGFaUqkEp3fT6jP854rVO2yQ==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/fetch-http-handler": "^5.6.4", - "@smithy/node-http-handler": "^4.9.4", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/fetch-http-handler": "^5.6.6", + "@smithy/node-http-handler": "^4.9.6", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -732,23 +722,23 @@ } }, "node_modules/@aws-sdk/credential-provider-ini": { - "version": "3.973.1", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-ini/-/credential-provider-ini-3.973.1.tgz", - "integrity": "sha512-6d8H6ZAh3ZPKZ6fe1nG2OWeZEZPtt9ravoD1dezPdPtsSkJRoxGAnFSHwKT3E/Te6fHE30zRzjV6TD12rvF6yQ==", + "version": "3.973.3", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-ini/-/credential-provider-ini-3.973.3.tgz", + "integrity": "sha512-WpuqYX4gGkx++fCTSWE8+41JzkZVcrI50SH48Ml4CsG1pyuHKyMmpw/FixBHDrmjoQ553PmeCLa/fZIcst+WyA==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/credential-provider-env": "^3.972.57", - "@aws-sdk/credential-provider-http": "^3.972.59", - "@aws-sdk/credential-provider-login": "^3.972.63", - "@aws-sdk/credential-provider-process": "^3.972.57", - "@aws-sdk/credential-provider-sso": "^3.973.1", - "@aws-sdk/credential-provider-web-identity": "^3.972.63", - "@aws-sdk/nested-clients": "^3.997.31", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/credential-provider-imds": "^4.4.7", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/credential-provider-env": "^3.972.59", + "@aws-sdk/credential-provider-http": "^3.972.61", + "@aws-sdk/credential-provider-login": "^3.972.65", + "@aws-sdk/credential-provider-process": "^3.972.59", + "@aws-sdk/credential-provider-sso": "^3.973.3", + "@aws-sdk/credential-provider-web-identity": "^3.972.65", + "@aws-sdk/nested-clients": "^3.997.33", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/credential-provider-imds": "^4.4.9", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -756,16 +746,16 @@ } }, "node_modules/@aws-sdk/credential-provider-login": { - "version": "3.972.63", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-login/-/credential-provider-login-3.972.63.tgz", - "integrity": "sha512-GREWRrMj0XnNKMaVa/Mauoaui26qBEHu71WWqXbwZOu/jFQOnPZjTf7u0KtGKC8VGa6VUs9kDWGgocrKNLS9vw==", + "version": "3.972.65", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-login/-/credential-provider-login-3.972.65.tgz", + "integrity": "sha512-xr9rgjYEdmC2Tpg2lwt9o+nOEaK9Qpd+dBjzrVCuWWyQfvhO91Ezu0Hh9ts2VUxOZxmS/k5T9msa34e4R1bnrQ==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/nested-clients": "^3.997.31", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/nested-clients": "^3.997.33", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -773,21 +763,21 @@ } }, "node_modules/@aws-sdk/credential-provider-node": { - "version": "3.972.67", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-node/-/credential-provider-node-3.972.67.tgz", - "integrity": "sha512-oYlzWst56rlhhjbYnexwv5hVLYe1cW4liLObhDfxDLI4RAQzleMVHQgQgx7XsC4HKj4e3kjT8v9DId+Pi/dndw==", + "version": "3.972.69", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-node/-/credential-provider-node-3.972.69.tgz", + "integrity": "sha512-wbJGGesd0Tl18bmUcbj1xJ+e7CpuRJ6PIpMywLFuUttGy615lua87cJ0EA8pFpY/QgPuUXbnupWBtSPJ9tyZhg==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/credential-provider-env": "^3.972.57", - "@aws-sdk/credential-provider-http": "^3.972.59", - "@aws-sdk/credential-provider-ini": "^3.973.1", - "@aws-sdk/credential-provider-process": "^3.972.57", - "@aws-sdk/credential-provider-sso": "^3.973.1", - "@aws-sdk/credential-provider-web-identity": "^3.972.63", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/credential-provider-imds": "^4.4.7", - "@smithy/types": "^4.16.0", + "@aws-sdk/credential-provider-env": "^3.972.59", + "@aws-sdk/credential-provider-http": "^3.972.61", + "@aws-sdk/credential-provider-ini": "^3.973.3", + "@aws-sdk/credential-provider-process": "^3.972.59", + "@aws-sdk/credential-provider-sso": "^3.973.3", + "@aws-sdk/credential-provider-web-identity": "^3.972.65", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/credential-provider-imds": "^4.4.9", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -795,15 +785,15 @@ } }, "node_modules/@aws-sdk/credential-provider-process": { - "version": "3.972.57", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-process/-/credential-provider-process-3.972.57.tgz", - "integrity": "sha512-TiVQhuU0pbhIZAUZacbPHMyzrIdiH+lnx+PMY/Pu/b93dJrq3wdZwzUJ0TPpvNxaqbHsxJvQZW3/h/beLiKq7Q==", + "version": "3.972.59", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-process/-/credential-provider-process-3.972.59.tgz", + "integrity": "sha512-DlZF2/MhLlatDdlrIy3CUCpfdbLrKx+3SMjVo+WyHnPpwzkc/M3vwAHw4OVJf7DMvO+4vfRqSCMc/E9I1auN0g==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -811,34 +801,17 @@ } }, "node_modules/@aws-sdk/credential-provider-sso": { - "version": "3.973.1", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-sso/-/credential-provider-sso-3.973.1.tgz", - "integrity": "sha512-3foTZUJ4821Ij60X7K3NJroygiZLnbBmarN+T//O2cjkISan90zElN3NBmgSlDrTQ7Gs6z/yO8V7h60QNcDZHQ==", + "version": "3.973.3", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-sso/-/credential-provider-sso-3.973.3.tgz", + "integrity": "sha512-hmdDHoy2G5Es2e8IgelNMYUuSQI6uCIAKZMJ2u2PdKDhxvbk1uWD/g4+R7R5c/tJfKEB1+KjjWiaoCr/S+ZTiQ==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/nested-clients": "^3.997.31", - "@aws-sdk/token-providers": "3.1083.0", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/types": "^4.16.0", - "tslib": "^2.6.2" - }, - "engines": { - "node": ">=20.0.0" - } - }, - "node_modules/@aws-sdk/credential-provider-sso/node_modules/@aws-sdk/token-providers": { - "version": "3.1083.0", - "resolved": "https://registry.npmjs.org/@aws-sdk/token-providers/-/token-providers-3.1083.0.tgz", - "integrity": "sha512-s0woKnxuHrExLc5L2ArIH5BMkbonHPtt+5hSBM8oknp9M6QTuUmmAmJ2E0EdzCGONrO+8+ADPqvv6UX0nNcc7A==", - "license": "Apache-2.0", - "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/nested-clients": "^3.997.31", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/nested-clients": "^3.997.33", + "@aws-sdk/token-providers": "3.1088.0", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -846,16 +819,16 @@ } }, "node_modules/@aws-sdk/credential-provider-web-identity": { - "version": "3.972.63", - "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-web-identity/-/credential-provider-web-identity-3.972.63.tgz", - "integrity": "sha512-8qZLFhM69eKcS37m459ctPR05Qimycm/74OPVioe6wNZabMT54GYhwBju0+J656RkMasNSawWQu+c8CmBe3TUQ==", + "version": "3.972.65", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-web-identity/-/credential-provider-web-identity-3.972.65.tgz", + "integrity": "sha512-gHQb/Kt0chjk/JQDa/GJDqmAvEuVn8n7z10wK2h0LFM9TUDRkohgOO4aEF+s2sBLM0br7Cl5W6P7phgjrrJvLQ==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/nested-clients": "^3.997.31", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/nested-clients": "^3.997.33", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -863,14 +836,14 @@ } }, "node_modules/@aws-sdk/eventstream-handler-node": { - "version": "3.972.25", - "resolved": "https://registry.npmjs.org/@aws-sdk/eventstream-handler-node/-/eventstream-handler-node-3.972.25.tgz", - "integrity": "sha512-df7HN1ozwMrB9+59re9PM7tSLxLAcheMWc5u/KyfCPCAWtN/vP7y7RTUZOy48uT1K9MESisVeOPPzF3O1AW01A==", + "version": "3.972.28", + "resolved": "https://registry.npmjs.org/@aws-sdk/eventstream-handler-node/-/eventstream-handler-node-3.972.28.tgz", + "integrity": "sha512-XV5sEH1xH5oydNgvUH87CR8BA2SBZXltckteS17D7ZT2k2THIBiExO9TEBYcgqUU31WAIfBH2hmx+fhFMiTL4Q==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/types": "^3.973.15", - "@smithy/core": "^3.29.0", - "@smithy/types": "^4.15.1", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -878,14 +851,14 @@ } }, "node_modules/@aws-sdk/middleware-eventstream": { - "version": "3.972.21", - "resolved": "https://registry.npmjs.org/@aws-sdk/middleware-eventstream/-/middleware-eventstream-3.972.21.tgz", - "integrity": "sha512-HvLgDnxBLaHi9E5K++6Vuk+1+qqn7Pmn8zrlzd+NXH3jBzwujnuzZtAR9WHPkbUGPO92FkoQWj/M1IsdxTlBmQ==", + "version": "3.972.24", + "resolved": "https://registry.npmjs.org/@aws-sdk/middleware-eventstream/-/middleware-eventstream-3.972.24.tgz", + "integrity": "sha512-oykin4mDWxNOuYQ7SF1cHzgYeuFEkF4cdRwgvjFFbIklkx09qIFBiOgsORafG9sXZFO3TayMmQuAQYgADXhI8w==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/types": "^3.973.15", - "@smithy/core": "^3.29.0", - "@smithy/types": "^4.15.1", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -912,17 +885,17 @@ } }, "node_modules/@aws-sdk/middleware-websocket": { - "version": "3.972.37", - "resolved": "https://registry.npmjs.org/@aws-sdk/middleware-websocket/-/middleware-websocket-3.972.37.tgz", - "integrity": "sha512-u4J2KwTe6hr0hBrcKF7vPNxoQoPdSwdhE8mEQK/ffaY/XgYQ77NRqsbxeNQDaRthQJ3D3KhOdCjrio1E+/ocng==", + "version": "3.972.41", + "resolved": "https://registry.npmjs.org/@aws-sdk/middleware-websocket/-/middleware-websocket-3.972.41.tgz", + "integrity": "sha512-LSbGvvYmjc4Br9BPYI2dTLnIclmrSiQbahkP4D6nRGVEv4qsCZ8csVuKBPVEEFCVD+EEngGh8ROls6XpumtwMg==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.974.29", - "@aws-sdk/types": "^3.973.15", - "@smithy/core": "^3.29.0", - "@smithy/fetch-http-handler": "^5.6.2", - "@smithy/signature-v4": "^5.6.1", - "@smithy/types": "^4.15.1", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/fetch-http-handler": "^5.6.6", + "@smithy/signature-v4": "^5.6.5", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -930,18 +903,18 @@ } }, "node_modules/@aws-sdk/nested-clients": { - "version": "3.997.31", - "resolved": "https://registry.npmjs.org/@aws-sdk/nested-clients/-/nested-clients-3.997.31.tgz", - "integrity": "sha512-BDHTpwcsZHEBNEJzOg/B1BkFYJxAXY50dau/NyVWs3d51F0WgIUGSWZot/Os+N3KpDhXeaXnz37mWffAvduREw==", + "version": "3.997.33", + "resolved": "https://registry.npmjs.org/@aws-sdk/nested-clients/-/nested-clients-3.997.33.tgz", + "integrity": "sha512-dVZOroI/r3/ENvqNGgjMPul+jjlz9GddfVusgTXlVjfZj5isibOxecLkGQbRPp8XOuX+RAfjXLFgPkD1JS5xrw==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.975.1", - "@aws-sdk/signature-v4-multi-region": "^3.996.39", - "@aws-sdk/types": "^3.974.0", - "@smithy/core": "^3.29.2", - "@smithy/fetch-http-handler": "^5.6.4", - "@smithy/node-http-handler": "^4.9.4", - "@smithy/types": "^4.16.0", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/signature-v4-multi-region": "^3.996.41", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/fetch-http-handler": "^5.6.6", + "@smithy/node-http-handler": "^4.9.6", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -949,14 +922,14 @@ } }, "node_modules/@aws-sdk/signature-v4-multi-region": { - "version": "3.996.39", - "resolved": "https://registry.npmjs.org/@aws-sdk/signature-v4-multi-region/-/signature-v4-multi-region-3.996.39.tgz", - "integrity": "sha512-8+srXqYIF8KYMLC4FxMLEM5Ek7kUNibJu1R4m8/fUhhNYIZZz26oGtKkCr8I/HiG2fFQxBvaGgQZT4/mqRCSnA==", + "version": "3.996.41", + "resolved": "https://registry.npmjs.org/@aws-sdk/signature-v4-multi-region/-/signature-v4-multi-region-3.996.41.tgz", + "integrity": "sha512-QMUytg+FQMGouc8gHS00KoYih3+N6cqmVI/pQGOIo7Nr7OpQaiXjSYOuL+vsPZ1tymY4LAQ8MYcHJmws5LRxng==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/types": "^3.974.0", - "@smithy/signature-v4": "^5.6.3", - "@smithy/types": "^4.16.0", + "@aws-sdk/types": "^3.974.2", + "@smithy/signature-v4": "^5.6.5", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -964,16 +937,16 @@ } }, "node_modules/@aws-sdk/token-providers": { - "version": "3.1081.0", - "resolved": "https://registry.npmjs.org/@aws-sdk/token-providers/-/token-providers-3.1081.0.tgz", - "integrity": "sha512-kduAeI6cL+zqwj3gjPh9LhuX7kBZ83msYxutavaR+UPm5K8J7iThJBvNRAsFNyWTji92CSU8dogUgvi9T0BehA==", + "version": "3.1088.0", + "resolved": "https://registry.npmjs.org/@aws-sdk/token-providers/-/token-providers-3.1088.0.tgz", + "integrity": "sha512-4ObatWt2qpJg5FBk4LOOKrTQYzaqeewAtdO3r9ZO8lH9YqLtpTzLyIdy0mJ+nVdfYOnqISkKNfmzP22bNDhwyw==", "license": "Apache-2.0", "dependencies": { - "@aws-sdk/core": "^3.974.29", - "@aws-sdk/nested-clients": "^3.997.29", - "@aws-sdk/types": "^3.973.15", - "@smithy/core": "^3.29.0", - "@smithy/types": "^4.15.1", + "@aws-sdk/core": "^3.975.3", + "@aws-sdk/nested-clients": "^3.997.33", + "@aws-sdk/types": "^3.974.2", + "@smithy/core": "^3.29.4", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -981,12 +954,12 @@ } }, "node_modules/@aws-sdk/types": { - "version": "3.974.0", - "resolved": "https://registry.npmjs.org/@aws-sdk/types/-/types-3.974.0.tgz", - "integrity": "sha512-QIBrw90CDm4O0UaIIzkU6DrFdeJzEb2Va5EPEVpyldj6sHJxB6cshhStJuhZxk3wR3PmjJlYsjPmY1kNb+KGBg==", + "version": "3.974.2", + "resolved": "https://registry.npmjs.org/@aws-sdk/types/-/types-3.974.2.tgz", + "integrity": "sha512-3W6IUtSxFbH6X7Wb7DzGCV5QiFQsd0g8bOfntpmDxQlzBoKWUMBu/JPQR0DwkE+Hpnxd6db1tXbOwdeHddG6cA==", "license": "Apache-2.0", "dependencies": { - "@smithy/types": "^4.16.0", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -994,12 +967,12 @@ } }, "node_modules/@aws-sdk/xml-builder": { - "version": "3.972.34", - "resolved": "https://registry.npmjs.org/@aws-sdk/xml-builder/-/xml-builder-3.972.34.tgz", - "integrity": "sha512-wHhWL1y7sN3enBA8POrPpQM5jCcmu2ozyhbRei4c8OjVcEaEs6yLucLa/pla457ggS/ysuy7bosagz3HaJkZXA==", + "version": "3.972.36", + "resolved": "https://registry.npmjs.org/@aws-sdk/xml-builder/-/xml-builder-3.972.36.tgz", + "integrity": "sha512-RdGmS1GLrtaTOLE1ElSluMldNrpk9Emq6uYs8SS8iHlu5xTAmM9rRkM91o48+rIRryBtyO9t+uLYCoMG6jVMVA==", "license": "Apache-2.0", "dependencies": { - "@smithy/types": "^4.16.0", + "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, "engines": { @@ -3071,9 +3044,9 @@ } }, "node_modules/@eslint/eslintrc": { - "version": "3.3.5", - "resolved": "https://registry.npmjs.org/@eslint/eslintrc/-/eslintrc-3.3.5.tgz", - "integrity": "sha512-4IlJx0X0qftVsN5E+/vGujTRIFtwuLbNsVUe7TO6zYPDR1O6nFwvwhIKEKSrl6dZchmYBITazxKoUYOjdtjlRg==", + "version": "3.3.6", + "resolved": "https://registry.npmjs.org/@eslint/eslintrc/-/eslintrc-3.3.6.tgz", + "integrity": "sha512-l2Ul9PrHsPCKcEY/ac7VgFj9D80C7S68sOKc618SyHDPK36s1XcFebXY0iTzUVn4Yq+YbwvSnDmCz9yxjX+QrA==", "dev": true, "license": "MIT", "dependencies": { @@ -3083,7 +3056,7 @@ "globals": "^14.0.0", "ignore": "^5.2.0", "import-fresh": "^3.2.1", - "js-yaml": "^4.1.1", + "js-yaml": "^4.3.0", "minimatch": "^3.1.5", "strip-json-comments": "^3.1.1" }, @@ -3112,9 +3085,9 @@ } }, "node_modules/@eslint/eslintrc/node_modules/js-yaml": { - "version": "4.2.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz", - "integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==", + "version": "4.3.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz", + "integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==", "dev": true, "funding": [ { @@ -3142,9 +3115,9 @@ "license": "MIT" }, "node_modules/@eslint/js": { - "version": "9.39.4", - "resolved": "https://registry.npmjs.org/@eslint/js/-/js-9.39.4.tgz", - "integrity": "sha512-nE7DEIchvtiFTwBw4Lfbu59PG+kCofhjsKaCWzxTpt4lfRjRMqG6uMBzKXuEcyXhOHoUp9riAm7/aWYGhXZ9cw==", + "version": "9.39.5", + "resolved": "https://registry.npmjs.org/@eslint/js/-/js-9.39.5.tgz", + "integrity": "sha512-QywQuszQh77pIXCsq998c8hbhSTI/azTty1Z6N53dmAudKHhy573j3yvRLsX2BSp8YpLtoCEG8E9DJe+8zUh4A==", "dev": true, "license": "MIT", "engines": { @@ -3268,18 +3241,18 @@ "license": "MIT" }, "node_modules/@formatjs/icu-messageformat-parser": { - "version": "3.5.12", - "resolved": "https://registry.npmjs.org/@formatjs/icu-messageformat-parser/-/icu-messageformat-parser-3.5.12.tgz", - "integrity": "sha512-YyzzxVgYJ8DELmmkhn0Yr0rUj0dTJFf9Jp628K3S0ysInBWxLVDOS8i3RP91cCp4DMK4WYb4cVMhWA9i4knSJg==", + "version": "3.5.14", + "resolved": "https://registry.npmjs.org/@formatjs/icu-messageformat-parser/-/icu-messageformat-parser-3.5.14.tgz", + "integrity": "sha512-jDvgtoLqe3U6yzoBlToMTkWBe38qSi7LN7kFlnXzd5ig8nn+4tSlED0xtEtdYakZVZGJJY2rW1D5xS3BFZh6kA==", "license": "MIT", "dependencies": { - "@formatjs/icu-skeleton-parser": "2.1.10" + "@formatjs/icu-skeleton-parser": "2.1.11" } }, "node_modules/@formatjs/icu-skeleton-parser": { - "version": "2.1.10", - "resolved": "https://registry.npmjs.org/@formatjs/icu-skeleton-parser/-/icu-skeleton-parser-2.1.10.tgz", - "integrity": "sha512-XuSva+8ZGawk8VnD5VD6UeH8KarQ/Z022zgjHDoHmlNiAewstXuuzXc0Hk5pGFSdG+nNw5bfJKXqj1ZXHn9yUA==", + "version": "2.1.11", + "resolved": "https://registry.npmjs.org/@formatjs/icu-skeleton-parser/-/icu-skeleton-parser-2.1.11.tgz", + "integrity": "sha512-j8cUmOJzVgkHuS0QiQ6ga76UIoLOFSAMWhs7aZJztH3aAdCOAE6vpC8KVvFB4cU10ON0y2/5oOVmPJ43s2lTwA==", "license": "MIT" }, "node_modules/@formatjs/intl-localematcher": { @@ -3308,9 +3281,9 @@ } }, "node_modules/@fumadocs/tailwind": { - "version": "0.1.0", - "resolved": "https://registry.npmjs.org/@fumadocs/tailwind/-/tailwind-0.1.0.tgz", - "integrity": "sha512-nF/DCAwOR21HZ4AkjIOv3Iqwyqywzb6pdyeMcoa+aZzirXj5ntvNZbe3jJ0v3ehhtrRfYYeXBezvjn8ZmV+fuQ==", + "version": "0.1.1", + "resolved": "https://registry.npmjs.org/@fumadocs/tailwind/-/tailwind-0.1.1.tgz", + "integrity": "sha512-BnPe52UxSaG8yKlHMKBxXw8h6GpK5qO55ci6+Qd5JnquTvIw6SpfbC1P+qAi82PuPWv1KZAWY8bxRk4+x9ctXw==", "license": "MIT", "peerDependencies": { "tailwindcss": "^4.0.0" @@ -3575,30 +3548,6 @@ "node": ">=20.0.0" } }, - "node_modules/@ibm-generative-ai/node-sdk": { - "version": "3.2.4", - "resolved": "https://registry.npmjs.org/@ibm-generative-ai/node-sdk/-/node-sdk-3.2.4.tgz", - "integrity": "sha512-HvJSYql3lOPYZcGb23mBw0kcWLlCX+n7EDRgJQxz7gIzx9WafUuDyl1IlTCXGfxolm0EhNIub79u9v7owtks0w==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "@ai-zen/node-fetch-event-source": "^2.1.2", - "fetch-retry": "^5.0.6", - "http-status-codes": "^2.3.0", - "openapi-fetch": "^0.8.2", - "p-queue-compat": "1.0.225", - "yaml": "^2.3.3" - }, - "peerDependencies": { - "@langchain/core": ">=0.1.0" - }, - "peerDependenciesMeta": { - "@langchain/core": { - "optional": true - } - } - }, "node_modules/@iconify/types": { "version": "2.0.0", "resolved": "https://registry.npmjs.org/@iconify/types/-/types-2.0.0.tgz", @@ -4994,16 +4943,16 @@ ] }, "node_modules/@lobehub/icons": { - "version": "5.10.1", - "resolved": "https://registry.npmjs.org/@lobehub/icons/-/icons-5.10.1.tgz", - "integrity": "sha512-KMaE+YqPAXuA8gcmzBFefLa9KgCqmJy9Mg3tlGedrL2coAzCQeps+aqivjejHNMnCDTPnGb+OHvX1um2kT1lQw==", + "version": "5.13.0", + "resolved": "https://registry.npmjs.org/@lobehub/icons/-/icons-5.13.0.tgz", + "integrity": "sha512-iXQF8GFvlwNJMR+PaU3jCgVAn5B8F7P48Fm6aodSXP+b+HJiR266rvlMSYvCULRAB/6/rtS1WZH3npc3p3viFw==", "license": "MIT", "workspaces": [ "packages/*" ], "dependencies": { "antd-style": "^4.1.0", - "es-toolkit": "^1.45.1", + "es-toolkit": "^1.49.0", "lucide-react": "^0.469.0", "polished": "^4.3.1" }, @@ -6514,9 +6463,9 @@ } }, "node_modules/@openai/codex": { - "version": "0.142.5", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5.tgz", - "integrity": "sha512-WQEpD7l3k68eIAP0aq28EdR18ENBAf8DyprzFhzNwCOQJSv4nHzpwT8Fl30IJacprko2ZCmUBZjM2u941l2yLw==", + "version": "0.144.4", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4.tgz", + "integrity": "sha512-DTHzYatlKq9dw55E0/HsbK4tRCEKabuJ10ybbqpsG8gVv/kvwEdg3Z4OI3cvLXKa21xkIa4lkGlZoO/HmqmFFw==", "dev": true, "license": "Apache-2.0", "optional": true, @@ -6527,19 +6476,19 @@ "node": ">=16" }, "optionalDependencies": { - "@openai/codex-darwin-arm64": "npm:@openai/codex@0.142.5-darwin-arm64", - "@openai/codex-darwin-x64": "npm:@openai/codex@0.142.5-darwin-x64", - "@openai/codex-linux-arm64": "npm:@openai/codex@0.142.5-linux-arm64", - "@openai/codex-linux-x64": "npm:@openai/codex@0.142.5-linux-x64", - "@openai/codex-win32-arm64": "npm:@openai/codex@0.142.5-win32-arm64", - "@openai/codex-win32-x64": "npm:@openai/codex@0.142.5-win32-x64" + "@openai/codex-darwin-arm64": "npm:@openai/codex@0.144.4-darwin-arm64", + "@openai/codex-darwin-x64": "npm:@openai/codex@0.144.4-darwin-x64", + "@openai/codex-linux-arm64": "npm:@openai/codex@0.144.4-linux-arm64", + "@openai/codex-linux-x64": "npm:@openai/codex@0.144.4-linux-x64", + "@openai/codex-win32-arm64": "npm:@openai/codex@0.144.4-win32-arm64", + "@openai/codex-win32-x64": "npm:@openai/codex@0.144.4-win32-x64" } }, "node_modules/@openai/codex-darwin-arm64": { "name": "@openai/codex", - "version": "0.142.5-darwin-arm64", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5-darwin-arm64.tgz", - "integrity": "sha512-l43p8xv+Z/2/b6fCUc7/FmcQZsaPB7RFizLponGwHAnFOWe3i9Vky69p+up3BUam9AetoQQUv7Mo+2KdaFEqhA==", + "version": "0.144.4-darwin-arm64", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4-darwin-arm64.tgz", + "integrity": "sha512-6J3g498cM2oA7vYIJhpuGJlnIi/M5JdYmjB5BZ1Of5HQ0ziIlplFSvH801oVy9J5TQFp642ODzOu/ZEokDUXsg==", "cpu": [ "arm64" ], @@ -6555,9 +6504,9 @@ }, "node_modules/@openai/codex-darwin-x64": { "name": "@openai/codex", - "version": "0.142.5-darwin-x64", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5-darwin-x64.tgz", - "integrity": "sha512-yk6A06/VmW7NFsa48OVPaj//g/zeSpd79wjuqfXZwW8ZKRYQm3+wCd3hWjPl79F3QnXvDvM2j3JMIBL3m3GXXg==", + "version": "0.144.4-darwin-x64", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4-darwin-x64.tgz", + "integrity": "sha512-k1HC8gdbAy+VmMbekYkhM+r+QE2Xfgd67n1VSp94tjz7aXVKoalHcDkdKNM/uUQ8o2tvbiwhHSUftJF8Sm9/Lw==", "cpu": [ "x64" ], @@ -6573,9 +6522,9 @@ }, "node_modules/@openai/codex-linux-arm64": { "name": "@openai/codex", - "version": "0.142.5-linux-arm64", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5-linux-arm64.tgz", - "integrity": "sha512-77ka5PSnm5HdxdBT99IwntCasmbqevlS0eiC0AtEb6ZXCLkim2gm0AWm+jNYy0EhbssvNK+KghayWo34HMgXeA==", + "version": "0.144.4-linux-arm64", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4-linux-arm64.tgz", + "integrity": "sha512-OlKx65579OwIzech9Tt3OUH9+hFZfFrCBP1hL2MudnMIoNr1+cFZjB5YIj5MWMRoBD+K5W3wdBIpQSH855b5Sg==", "cpu": [ "arm64" ], @@ -6591,9 +6540,9 @@ }, "node_modules/@openai/codex-linux-x64": { "name": "@openai/codex", - "version": "0.142.5-linux-x64", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5-linux-x64.tgz", - "integrity": "sha512-pxY+d3NgNE57Y/MApD3/TZUAygxJN6I9h3ZeDUwe67mxWjUxsuapxMRFTKSznCalYbRAeZp752+AAXmUbmguEg==", + "version": "0.144.4-linux-x64", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4-linux-x64.tgz", + "integrity": "sha512-2jxrmV6+/7eBNdg5uhhmOEPFu2o28eYY/ClLzWhSBHH8uo3f2KA1z9JQcVtwlbToW03nEPlEzYNYfCF1UBqsVQ==", "cpu": [ "x64" ], @@ -6608,14 +6557,14 @@ } }, "node_modules/@openai/codex-sdk": { - "version": "0.142.5", - "resolved": "https://registry.npmjs.org/@openai/codex-sdk/-/codex-sdk-0.142.5.tgz", - "integrity": "sha512-MConZ+eoBoZmkc4reezuzOgLtoI1BQBzo/nVYsSjtAIBpwKcgeEm1rfmqfUnTfFaBNHFTxBntcS7ZeQYuDPbWA==", + "version": "0.144.4", + "resolved": "https://registry.npmjs.org/@openai/codex-sdk/-/codex-sdk-0.144.4.tgz", + "integrity": "sha512-JtND4npLaM1jKlTJVGU3XaX7xqnRG62yusz13kPgialnfd0CBrjOgXFGSRojYcvWZSggQoEkZBVAGtoKE0y8sw==", "dev": true, "license": "Apache-2.0", "optional": true, "dependencies": { - "@openai/codex": "0.142.5" + "@openai/codex": "0.144.4" }, "engines": { "node": ">=18" @@ -6623,9 +6572,9 @@ }, "node_modules/@openai/codex-win32-arm64": { "name": "@openai/codex", - "version": "0.142.5-win32-arm64", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5-win32-arm64.tgz", - "integrity": "sha512-65BEqGbUZ7r0ayunIHdBjo5crwgbwKX/6puOcO+VCswUw/dXvDsN2IGcbXB52+bS9U5+FxP783cUHfTT6m40DQ==", + "version": "0.144.4-win32-arm64", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4-win32-arm64.tgz", + "integrity": "sha512-CCgfI1smFhHZTIpTuBwDJwBr/AR40RTqaFxbBWVabu0RMeYDteRuPiDfdTlktf3C43Y1q10VZXhVGYtCokDg2w==", "cpu": [ "arm64" ], @@ -6641,9 +6590,9 @@ }, "node_modules/@openai/codex-win32-x64": { "name": "@openai/codex", - "version": "0.142.5-win32-x64", - "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.142.5-win32-x64.tgz", - "integrity": "sha512-a+wI4PEx9a2fg6V5ueTTDkOkr1XpEvA5RFXIbo/L2hOfzMmGtyRnbG24bCGu5Q2RSgVxSQV0aLkdb3vdYMNH9A==", + "version": "0.144.4-win32-x64", + "resolved": "https://registry.npmjs.org/@openai/codex/-/codex-0.144.4-win32-x64.tgz", + "integrity": "sha512-iL1ky0ERgdQJOKzom/Ms1fhpwkSmpsA9eVrzAqURFlYGS8z7JqwEgm33+nLGCsY7y25d8Xs/LJ91Oiqz3yXcUg==", "cpu": [ "x64" ], @@ -6679,9 +6628,9 @@ } }, "node_modules/@opentelemetry/api-logs": { - "version": "0.219.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/api-logs/-/api-logs-0.219.0.tgz", - "integrity": "sha512-FFx7YnaYJlIjqWW/AG/yAZ0L/NEY724PipXXXQLdtZPbLwBGbUMTGL1i/esI56TWfTUXxhLfpgrnWJCG8aUJyg==", + "version": "0.220.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/api-logs/-/api-logs-0.220.0.tgz", + "integrity": "sha512-CmVa4ImJ+ynfrPMNaAXHET6Bhb44SwzmfyVJFq9ni2jgXJR/l7C6gfVFddNmHP+ZOkP9cf4f9DBe68qVLTHc9w==", "dev": true, "license": "Apache-2.0", "dependencies": { @@ -6705,9 +6654,9 @@ } }, "node_modules/@opentelemetry/core": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.8.0.tgz", - "integrity": "sha512-hd1Lfh8p545nNz+jq1Ejfz+Mn1hyLuxYn1YzTfFNrxr8urEWMNQLPf1Th8kjOH+HxwawCrtgBp8JpBUR4ZSgww==", + "version": "2.9.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.9.0.tgz", + "integrity": "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw==", "dev": true, "license": "Apache-2.0", "dependencies": { @@ -6721,17 +6670,17 @@ } }, "node_modules/@opentelemetry/exporter-trace-otlp-http": { - "version": "0.219.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/exporter-trace-otlp-http/-/exporter-trace-otlp-http-0.219.0.tgz", - "integrity": "sha512-9t6SvBXXBEjOBcIzgozvBbd3jWrv3Gt3ngGhl1fhdZ/zRc7oZDVOFEqbi2zlBpW9BXhgDMKv422J0DL/3iQWfw==", + "version": "0.220.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/exporter-trace-otlp-http/-/exporter-trace-otlp-http-0.220.0.tgz", + "integrity": "sha512-/+ExB3lRkf+erv4PnoywyL7RHKITidxtUpUTS55k7OQ0dB42S7gEF1gry7swb9MSm1hYLUhJg4QQh9W8SpwwqA==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/otlp-exporter-base": "0.219.0", - "@opentelemetry/otlp-transformer": "0.219.0", - "@opentelemetry/resources": "2.8.0", - "@opentelemetry/sdk-trace-base": "2.8.0" + "@opentelemetry/core": "2.9.0", + "@opentelemetry/otlp-exporter-base": "0.220.0", + "@opentelemetry/otlp-transformer": "0.220.0", + "@opentelemetry/resources": "2.9.0", + "@opentelemetry/sdk-trace": "2.9.0" }, "engines": { "node": "^18.19.0 || >=20.6.0" @@ -6740,50 +6689,15 @@ "@opentelemetry/api": "^1.3.0" } }, - "node_modules/@opentelemetry/exporter-trace-otlp-http/node_modules/@opentelemetry/resources": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.8.0.tgz", - "integrity": "sha512-qmXQ27ilDbUK/vGMqwL8D4/rhn76C+sherM4wTbjlfknR8Nvfc/hCxjRJPhkzZzUsPiNg16SA31NxMabwttRjg==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.3.0 <1.10.0" - } - }, - "node_modules/@opentelemetry/exporter-trace-otlp-http/node_modules/@opentelemetry/sdk-trace-base": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace-base/-/sdk-trace-base-2.8.0.tgz", - "integrity": "sha512-mhU4jp+vW0mGbFRd+GeXHvmfA4aDqWjBjLC3pE5XMpLs0IE2ryYb019Ts2AQrOq67gaTF25D91+fgvEHDZEnuQ==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/resources": "2.8.0", - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.3.0 <1.10.0" - } - }, "node_modules/@opentelemetry/otlp-exporter-base": { - "version": "0.219.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/otlp-exporter-base/-/otlp-exporter-base-0.219.0.tgz", - "integrity": "sha512-zvIxQX/AZUVKDU+hCuYx+7UkiP7GRdnk1ZbFQRYzHvYp47cAWR4j3IhoPhV9KaeXEv2xdGq3IA6PnpzDmLcmSA==", + "version": "0.220.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/otlp-exporter-base/-/otlp-exporter-base-0.220.0.tgz", + "integrity": "sha512-CXYo8UD5Mn9YbgebO2EL4wejtA+gxLmLiu6HCk2KH2BR7XhFN6/6p1UlCb23DYCjeYkndevLHuejCCN1yx4+OQ==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/otlp-transformer": "0.219.0" + "@opentelemetry/core": "2.9.0", + "@opentelemetry/otlp-transformer": "0.220.0" }, "engines": { "node": "^18.19.0 || >=20.6.0" @@ -6793,18 +6707,18 @@ } }, "node_modules/@opentelemetry/otlp-transformer": { - "version": "0.219.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/otlp-transformer/-/otlp-transformer-0.219.0.tgz", - "integrity": "sha512-aaYKAyXhw9VchKZVGOopD3Gw/kPsyrX2c6IQ0AW32mTjqmZOh5Y6Gf5OYqTNqVktAeBjmFinhyFaCwW6GYK9YQ==", + "version": "0.220.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/otlp-transformer/-/otlp-transformer-0.220.0.tgz", + "integrity": "sha512-lXGrv7KXZ0gNH9SVNUaa6vv6phVYGvJxfXAlMbzbakiXru75f5MZl8Z7oqiMMQD77riVHJCFlQvbZs/VVN2/4A==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@opentelemetry/api-logs": "0.219.0", - "@opentelemetry/core": "2.8.0", - "@opentelemetry/resources": "2.8.0", - "@opentelemetry/sdk-logs": "0.219.0", - "@opentelemetry/sdk-metrics": "2.8.0", - "@opentelemetry/sdk-trace-base": "2.8.0" + "@opentelemetry/api-logs": "0.220.0", + "@opentelemetry/core": "2.9.0", + "@opentelemetry/resources": "2.9.0", + "@opentelemetry/sdk-logs": "0.220.0", + "@opentelemetry/sdk-metrics": "2.9.0", + "@opentelemetry/sdk-trace": "2.9.0" }, "engines": { "node": "^18.19.0 || >=20.6.0" @@ -6813,41 +6727,6 @@ "@opentelemetry/api": "^1.3.0" } }, - "node_modules/@opentelemetry/otlp-transformer/node_modules/@opentelemetry/resources": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.8.0.tgz", - "integrity": "sha512-qmXQ27ilDbUK/vGMqwL8D4/rhn76C+sherM4wTbjlfknR8Nvfc/hCxjRJPhkzZzUsPiNg16SA31NxMabwttRjg==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.3.0 <1.10.0" - } - }, - "node_modules/@opentelemetry/otlp-transformer/node_modules/@opentelemetry/sdk-trace-base": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace-base/-/sdk-trace-base-2.8.0.tgz", - "integrity": "sha512-mhU4jp+vW0mGbFRd+GeXHvmfA4aDqWjBjLC3pE5XMpLs0IE2ryYb019Ts2AQrOq67gaTF25D91+fgvEHDZEnuQ==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/resources": "2.8.0", - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.3.0 <1.10.0" - } - }, "node_modules/@opentelemetry/resources": { "version": "2.9.0", "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.9.0.tgz", @@ -6865,32 +6744,16 @@ "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, - "node_modules/@opentelemetry/resources/node_modules/@opentelemetry/core": { - "version": "2.9.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.9.0.tgz", - "integrity": "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.0.0 <1.10.0" - } - }, "node_modules/@opentelemetry/sdk-logs": { - "version": "0.219.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-logs/-/sdk-logs-0.219.0.tgz", - "integrity": "sha512-s6lTKRakaPClvKoWHRChxnXjDMkM/TQ30ff78jN6EBGf7MI7VzANE5PU3f4z9qDUudWjvZjOLHG0rBnBKYvoXA==", + "version": "0.220.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-logs/-/sdk-logs-0.220.0.tgz", + "integrity": "sha512-WywcTkQtv2iNmt+6y5Kcd4rzvx9bLVsBa2Nwcmg01IUaBTkTow3W4d9KE5vNBpEDtb9tp21WcRBY/lANRrApYA==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@opentelemetry/api-logs": "0.219.0", - "@opentelemetry/core": "2.8.0", - "@opentelemetry/resources": "2.8.0", + "@opentelemetry/api-logs": "0.220.0", + "@opentelemetry/core": "2.9.0", + "@opentelemetry/resources": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "engines": { @@ -6900,32 +6763,15 @@ "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, - "node_modules/@opentelemetry/sdk-logs/node_modules/@opentelemetry/resources": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.8.0.tgz", - "integrity": "sha512-qmXQ27ilDbUK/vGMqwL8D4/rhn76C+sherM4wTbjlfknR8Nvfc/hCxjRJPhkzZzUsPiNg16SA31NxMabwttRjg==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.3.0 <1.10.0" - } - }, "node_modules/@opentelemetry/sdk-metrics": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-metrics/-/sdk-metrics-2.8.0.tgz", - "integrity": "sha512-UDBGaj6W0Rgy5rTTaoxs8gVGF/aGkAKyjurJv7se6wjRxJu7FoquTLT/vt54DZfo4crbprYfhX/SOK9+BPw1qg==", + "version": "2.9.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-metrics/-/sdk-metrics-2.9.0.tgz", + "integrity": "sha512-Xx8RGS4H5XEBl01WuCreMIpiah9cCXMbSkeuIePPdD2cUpq/vUzYmj8E/MK1OsbOc93FuAD4jfn2WOacKwLn7Q==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/resources": "2.8.0" + "@opentelemetry/core": "2.9.0", + "@opentelemetry/resources": "2.9.0" }, "engines": { "node": "^18.19.0 || >=20.6.0" @@ -6934,23 +6780,6 @@ "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, - "node_modules/@opentelemetry/sdk-metrics/node_modules/@opentelemetry/resources": { - "version": "2.8.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.8.0.tgz", - "integrity": "sha512-qmXQ27ilDbUK/vGMqwL8D4/rhn76C+sherM4wTbjlfknR8Nvfc/hCxjRJPhkzZzUsPiNg16SA31NxMabwttRjg==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/core": "2.8.0", - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.3.0 <1.10.0" - } - }, "node_modules/@opentelemetry/sdk-trace": { "version": "2.9.0", "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace/-/sdk-trace-2.9.0.tgz", @@ -6988,22 +6817,6 @@ "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, - "node_modules/@opentelemetry/sdk-trace-base/node_modules/@opentelemetry/core": { - "version": "2.9.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.9.0.tgz", - "integrity": "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.0.0 <1.10.0" - } - }, "node_modules/@opentelemetry/sdk-trace-node": { "version": "2.9.0", "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace-node/-/sdk-trace-node-2.9.0.tgz", @@ -7022,38 +6835,6 @@ "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, - "node_modules/@opentelemetry/sdk-trace-node/node_modules/@opentelemetry/core": { - "version": "2.9.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.9.0.tgz", - "integrity": "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.0.0 <1.10.0" - } - }, - "node_modules/@opentelemetry/sdk-trace/node_modules/@opentelemetry/core": { - "version": "2.9.0", - "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.9.0.tgz", - "integrity": "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw==", - "dev": true, - "license": "Apache-2.0", - "dependencies": { - "@opentelemetry/semantic-conventions": "^1.29.0" - }, - "engines": { - "node": "^18.19.0 || >=20.6.0" - }, - "peerDependencies": { - "@opentelemetry/api": ">=1.0.0 <1.10.0" - } - }, "node_modules/@opentelemetry/semantic-conventions": { "version": "1.43.0", "resolved": "https://registry.npmjs.org/@opentelemetry/semantic-conventions/-/semantic-conventions-1.43.0.tgz", @@ -10117,9 +9898,9 @@ } }, "node_modules/@smithy/core": { - "version": "3.29.3", - "resolved": "https://registry.npmjs.org/@smithy/core/-/core-3.29.3.tgz", - "integrity": "sha512-L+Ys6ecjk5vwPMAKHBpPKlJ3DkqwNcnfEISXBZIsVvWG/XKXfsAP8mwIYlTeLcd2ElHdesPI8OuOmJSFAPhm6A==", + "version": "3.29.4", + "resolved": "https://registry.npmjs.org/@smithy/core/-/core-3.29.4.tgz", + "integrity": "sha512-G1GRglAabzEhqghJMBAd54FkRS7SAFGHEwbhcI9r+O+LIMuFsLyXkLZkCoFSgAglRu8s/URVXJB0hglq3ZipIg==", "license": "Apache-2.0", "dependencies": { "@smithy/types": "^4.16.1", @@ -10130,12 +9911,12 @@ } }, "node_modules/@smithy/credential-provider-imds": { - "version": "4.4.8", - "resolved": "https://registry.npmjs.org/@smithy/credential-provider-imds/-/credential-provider-imds-4.4.8.tgz", - "integrity": "sha512-q9J7JTiXrAhB8sDp4px97uEPT7CwKH61Co78grdNQvU8QZAdiuaSRhP0tUVf2ogy36RZTrlMU1rBmDEH+cnkiA==", + "version": "4.4.9", + "resolved": "https://registry.npmjs.org/@smithy/credential-provider-imds/-/credential-provider-imds-4.4.9.tgz", + "integrity": "sha512-2nfV4qRKiYeXU4zD2vvSCfg5dfp/BuhrM73vt7q9gzBhxs4rbPxXY21wo+kyI3bRmXcEGRnCLTaW8O437jzHIg==", "license": "Apache-2.0", "dependencies": { - "@smithy/core": "^3.29.3", + "@smithy/core": "^3.29.4", "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, @@ -10144,12 +9925,12 @@ } }, "node_modules/@smithy/fetch-http-handler": { - "version": "5.6.5", - "resolved": "https://registry.npmjs.org/@smithy/fetch-http-handler/-/fetch-http-handler-5.6.5.tgz", - "integrity": "sha512-SuqeisTyPoiIPtIYru/sGxGyXzmZ+8nnFOhC+qRPglt06Ebd1yH//CDltZB2J/3WBNVhwfUaZ0EtHB3cm2X32g==", + "version": "5.6.6", + "resolved": "https://registry.npmjs.org/@smithy/fetch-http-handler/-/fetch-http-handler-5.6.6.tgz", + "integrity": "sha512-NHLgAlORUFZjn5ZfhYuyyKMlXA1WLYOdGxEhyNxrPpbJzoacGbl0chn1lN2KiZ8mpNVk0tV5607CSYlYs/OFgw==", "license": "Apache-2.0", "dependencies": { - "@smithy/core": "^3.29.3", + "@smithy/core": "^3.29.4", "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, @@ -10158,12 +9939,12 @@ } }, "node_modules/@smithy/node-http-handler": { - "version": "4.9.5", - "resolved": "https://registry.npmjs.org/@smithy/node-http-handler/-/node-http-handler-4.9.5.tgz", - "integrity": "sha512-bNqdxTQTxmLbomSmlkZFz8L6B/feQ2HHzw4L2zY7Ecp2XffYAZq2uzdWDdxJHJFbEvqd+SRuluJso0P8+xPdbw==", + "version": "4.9.6", + "resolved": "https://registry.npmjs.org/@smithy/node-http-handler/-/node-http-handler-4.9.6.tgz", + "integrity": "sha512-odd+HYx3OLcXRSEz0ZeF3JQdSYdK8QnRgA2N87cPW7coWIbKfRk7a9VQjfeWQLqnzrDLk23KMEn46p8N7M/JFg==", "license": "Apache-2.0", "dependencies": { - "@smithy/core": "^3.29.3", + "@smithy/core": "^3.29.4", "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, @@ -10172,12 +9953,12 @@ } }, "node_modules/@smithy/signature-v4": { - "version": "5.6.4", - "resolved": "https://registry.npmjs.org/@smithy/signature-v4/-/signature-v4-5.6.4.tgz", - "integrity": "sha512-B89bpf2t/y/wia6LZ+4JfHXYQT9PnVftsH05rgJKKIStS7r/4XSs9HOjtPoLtgcA6HCW9jVqX5DBbq7E0PAkiQ==", + "version": "5.6.5", + "resolved": "https://registry.npmjs.org/@smithy/signature-v4/-/signature-v4-5.6.5.tgz", + "integrity": "sha512-MO5VEhwVl0BN7xVoVeNrZfiUFoQtqxUbgl6/RwOTlMMxCSjblG8twSrVTwz3J4w9WZxd2rBfBAUXjH77agspBg==", "license": "Apache-2.0", "dependencies": { - "@smithy/core": "^3.29.3", + "@smithy/core": "^3.29.4", "@smithy/types": "^4.16.1", "tslib": "^2.6.2" }, @@ -11981,9 +11762,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.1.0", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.1.0.tgz", - "integrity": "sha512-O0A1G3xPGy4w7AgQdAQYUlQ+BKk2Oovw8eRpofyp5KdBZULnbe+WqaOVNrm705SHphCiG4XHsACrSmPu1f+Kgw==", + "version": "26.1.1", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.1.1.tgz", + "integrity": "sha512-nxAkRSVkN1Y0JC1W8ky/fTfkGsMmcrRsbx+3XoZE+rMOX71kLYTV7fLXpqud1GpbpP5TuffXFqfX7fH2GgZREw==", "devOptional": true, "license": "MIT", "dependencies": { @@ -12155,17 +11936,17 @@ } }, "node_modules/@typescript-eslint/eslint-plugin": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/eslint-plugin/-/eslint-plugin-8.63.0.tgz", - "integrity": "sha512-rvwSgqT+DHpWdzfSzPatRLm02a0GlESt++9iy3hLCDY4BgkaLcl8LBi9Yh7XGFBpwcBE/K3024QuXWTpbz4FfQ==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/eslint-plugin/-/eslint-plugin-8.64.0.tgz", + "integrity": "sha512-CGvQPBxN3wZLu6Rz2kFUpZeoCm78xUic92ck39KPePkO1NPOwjCqdQnm5Q87tpWw9vcBvW8XLrDXjH9PWYtJ3Q==", "dev": true, "license": "MIT", "dependencies": { "@eslint-community/regexpp": "^4.12.2", - "@typescript-eslint/scope-manager": "8.63.0", - "@typescript-eslint/type-utils": "8.63.0", - "@typescript-eslint/utils": "8.63.0", - "@typescript-eslint/visitor-keys": "8.63.0", + "@typescript-eslint/scope-manager": "8.64.0", + "@typescript-eslint/type-utils": "8.64.0", + "@typescript-eslint/utils": "8.64.0", + "@typescript-eslint/visitor-keys": "8.64.0", "ignore": "^7.0.5", "natural-compare": "^1.4.0", "ts-api-utils": "^2.5.0" @@ -12178,15 +11959,15 @@ "url": "https://opencollective.com/typescript-eslint" }, "peerDependencies": { - "@typescript-eslint/parser": "^8.63.0", + "@typescript-eslint/parser": "^8.64.0", "eslint": "^8.57.0 || ^9.0.0 || ^10.0.0", "typescript": ">=4.8.4 <6.1.0" } }, "node_modules/@typescript-eslint/eslint-plugin/node_modules/ignore": { - "version": "7.0.5", - "resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.5.tgz", - "integrity": "sha512-Hs59xBNfUIunMFgWAbGX5cq6893IbWg4KnrjbYwX3tx0ztorVgTDA6B2sxf8ejHJ4wz8BqGUMYlnzNBer5NvGg==", + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.6.tgz", + "integrity": "sha512-BAg6QkE8W+TuQLrrw0Ugr7HegXduRuuj8/ti2kSOc+jz1dmx8/WNcjr6XGnq5YpDWxFwwaavqD0+jIUOKelTsw==", "dev": true, "license": "MIT", "engines": { @@ -12194,16 +11975,16 @@ } }, "node_modules/@typescript-eslint/parser": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/parser/-/parser-8.63.0.tgz", - "integrity": "sha512-gwh4gvvlaVDKKxyfxMG+Gnu1u9X0OQBwyGLkbwB65dIzBKnxeRiJlNFqlI3zwVhNXJIs6qV7mlFCn/BIajlVig==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/parser/-/parser-8.64.0.tgz", + "integrity": "sha512-KA0OshtlcCCXmbfqyZkM5pV3/WNraJf7DkJRLpyrmwPtud57H5BDX7C3k0LPSPxpprfRL+cJDGabF10mvNCoCw==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/scope-manager": "8.63.0", - "@typescript-eslint/types": "8.63.0", - "@typescript-eslint/typescript-estree": "8.63.0", - "@typescript-eslint/visitor-keys": "8.63.0", + "@typescript-eslint/scope-manager": "8.64.0", + "@typescript-eslint/types": "8.64.0", + "@typescript-eslint/typescript-estree": "8.64.0", + "@typescript-eslint/visitor-keys": "8.64.0", "debug": "^4.4.3" }, "engines": { @@ -12219,14 +12000,14 @@ } }, "node_modules/@typescript-eslint/project-service": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/project-service/-/project-service-8.63.0.tgz", - "integrity": "sha512-e5dh0/UI0ok53AlZ5wRkXCB32z/f2jUZqPR/ygAw5WYaSw8j9EoJWlS7wQjr/dmOaqWjnPIn2m+HhVPCMWGZVQ==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/project-service/-/project-service-8.64.0.tgz", + "integrity": "sha512-tk4WpOJ6IEbGrVHaNmM0YRrwAD3exZlIK3iadQNAxh4YKk6jvUQ4ecq18n+v7+meh+cJ3j+D8nbk8sRKhlwLQg==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/tsconfig-utils": "^8.63.0", - "@typescript-eslint/types": "^8.63.0", + "@typescript-eslint/tsconfig-utils": "^8.64.0", + "@typescript-eslint/types": "^8.64.0", "debug": "^4.4.3" }, "engines": { @@ -12241,14 +12022,14 @@ } }, "node_modules/@typescript-eslint/scope-manager": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/scope-manager/-/scope-manager-8.63.0.tgz", - "integrity": "sha512-uUyfMWCnDSN8bCpcrY8nGP2BLkQ9Xn0GsipcONcpIDWhwhO4ZSyHvyS14U3X75mzxWxL3I2UZIrenTzdzcJO8A==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/scope-manager/-/scope-manager-8.64.0.tgz", + "integrity": "sha512-CXEaFdYXjSTgKhisNkwCcJwTP8Pl+fmRrEQrri4nm3vU743bALrxzLmq7fHG/7e6a5xO0lDYeURpZmBuhHk54w==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.63.0", - "@typescript-eslint/visitor-keys": "8.63.0" + "@typescript-eslint/types": "8.64.0", + "@typescript-eslint/visitor-keys": "8.64.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -12259,9 +12040,9 @@ } }, "node_modules/@typescript-eslint/tsconfig-utils": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/tsconfig-utils/-/tsconfig-utils-8.63.0.tgz", - "integrity": "sha512-sUAbkulqBAsncKnbRP3+7CtQFRKicexnj7ZwNC6ddCR7EmrXvjvdCYMJbUIqMd6lwoEriZjwLo08aS5tSjVMHg==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/tsconfig-utils/-/tsconfig-utils-8.64.0.tgz", + "integrity": "sha512-2yo8rRNKuzbVWQp5kslhANqZ2uDAeROQHBRZNPu8JDsHmeFNj/XJJhX/FhNUWmkHHvoNsKa6+tHJiig87EzsQw==", "dev": true, "license": "MIT", "engines": { @@ -12276,15 +12057,15 @@ } }, "node_modules/@typescript-eslint/type-utils": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/type-utils/-/type-utils-8.63.0.tgz", - "integrity": "sha512-Nzzh/OGxVCOjObjaj1CQF2RUasyYy2Jfuh+zZ3PjLzG2fYRriAiZLib9UKtO+CpQAS3YHiAS+ckZDclwqI1TPA==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/type-utils/-/type-utils-8.64.0.tgz", + "integrity": "sha512-XWG4Fmmv/6SvyS9nH8jWrKs6terwJvE8cyRt1CzYYqzp9OrPhCT4cMc/f7C6RZCwG+qMmiffJS1/qJP8G1URtg==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.63.0", - "@typescript-eslint/typescript-estree": "8.63.0", - "@typescript-eslint/utils": "8.63.0", + "@typescript-eslint/types": "8.64.0", + "@typescript-eslint/typescript-estree": "8.64.0", + "@typescript-eslint/utils": "8.64.0", "debug": "^4.4.3", "ts-api-utils": "^2.5.0" }, @@ -12301,9 +12082,9 @@ } }, "node_modules/@typescript-eslint/types": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/types/-/types-8.63.0.tgz", - "integrity": "sha512-xyLtl9DUBBFrcJS4x2pIqGLH68/tC2uOa4Z7pUteW09D3bXnnXUom4dyPikzWgB7llmIc1zoeI3aoUdC4rPK/Q==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/types/-/types-8.64.0.tgz", + "integrity": "sha512-qjhfuTfLXjA4IOzXvz0rTjT01BqEiIgPoUeMwiEjnaHKJMTNo8rH5pYW1a2L/0Dnux2fPC85AeyJoWaGa8WxTA==", "dev": true, "license": "MIT", "engines": { @@ -12315,16 +12096,16 @@ } }, "node_modules/@typescript-eslint/typescript-estree": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/typescript-estree/-/typescript-estree-8.63.0.tgz", - "integrity": "sha512-ygBkU+B7ex5UI/gKhaqexWev79uISfIv7XQCRNYO/jmD8rGLPyWLAb3KMRT6nd8Gt9bmUBi9+iX6tBdYfOY81Q==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/typescript-estree/-/typescript-estree-8.64.0.tgz", + "integrity": "sha512-Pztpsn1aCE1oWDvDEfUk31nngvvF7vUB5SwHFEaZIFpvw7WJtqUHHL4plBZDA9HfWJJjL13BdG0YrJInTUvoVA==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/project-service": "8.63.0", - "@typescript-eslint/tsconfig-utils": "8.63.0", - "@typescript-eslint/types": "8.63.0", - "@typescript-eslint/visitor-keys": "8.63.0", + "@typescript-eslint/project-service": "8.64.0", + "@typescript-eslint/tsconfig-utils": "8.64.0", + "@typescript-eslint/types": "8.64.0", + "@typescript-eslint/visitor-keys": "8.64.0", "debug": "^4.4.3", "minimatch": "^10.2.2", "semver": "^7.7.3", @@ -12395,16 +12176,16 @@ } }, "node_modules/@typescript-eslint/utils": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/utils/-/utils-8.63.0.tgz", - "integrity": "sha512-fUKaeAvrTuQg/Tgt3nliAUSZHJM6DlCcfyEmxCvlX8kieWSStBX+5O5Fnidtc3i2JrH+9c/GL4RY2iasd/GPTA==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/utils/-/utils-8.64.0.tgz", + "integrity": "sha512-aJUGVB3+U0htrrCjoA8qukw8cm8fNCGAxK/tVoS70k8aeb7DETKeFozRiVFIwEeN9WJLsjaP3ph8I60tY2XZoQ==", "dev": true, "license": "MIT", "dependencies": { "@eslint-community/eslint-utils": "^4.9.1", - "@typescript-eslint/scope-manager": "8.63.0", - "@typescript-eslint/types": "8.63.0", - "@typescript-eslint/typescript-estree": "8.63.0" + "@typescript-eslint/scope-manager": "8.64.0", + "@typescript-eslint/types": "8.64.0", + "@typescript-eslint/typescript-estree": "8.64.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -12419,13 +12200,13 @@ } }, "node_modules/@typescript-eslint/visitor-keys": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/visitor-keys/-/visitor-keys-8.63.0.tgz", - "integrity": "sha512-UexrHGnGTpbuQHct2ExOc2ZcFbGUS9FOesCxxqdBGcpI1BxYu/LZ6U8Aq6/72XtF/qRBk9nhuGHFJIXXMhPMdw==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/visitor-keys/-/visitor-keys-8.64.0.tgz", + "integrity": "sha512-mrtuL8Nsn6gi2H4mo5KMTp823M+3Q19Ew/i+Zlikq20tIMm99C3Ez0dCmkWWnxut20esQvTg8aUSEhMcAOXhEw==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.63.0", + "@typescript-eslint/types": "8.64.0", "eslint-visitor-keys": "^5.0.0" }, "engines": { @@ -13056,6 +12837,162 @@ "js-yaml": "bin/js-yaml.js" } }, + "node_modules/@yuku-analyzer/binding-darwin-arm64": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-darwin-arm64/-/binding-darwin-arm64-0.6.3.tgz", + "integrity": "sha512-1PI1tdfk0ozQ0tbEi740fYMz/3axKn+jR2nK2qBXdYZiyQKsPYW7lDockNbUY9Z5E3+nwEFjX6Pp19X4VIgrkQ==", + "cpu": [ + "arm64" + ], + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@yuku-analyzer/binding-darwin-x64": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-darwin-x64/-/binding-darwin-x64-0.6.3.tgz", + "integrity": "sha512-VyC+KH0gwPzXjtysXbuBop+Qn107800pQhp8YzDElnBciu/X88Uw3xEJrCJtcyoV85sPpq+g1zvAgHWHwE8PlQ==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@yuku-analyzer/binding-freebsd-x64": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-freebsd-x64/-/binding-freebsd-x64-0.6.3.tgz", + "integrity": "sha512-T5HRWQiy0e5bHaI01xn3cguopn0YCvNV4rast6p+o4ZjORLguCwM7i3GujOUXMzwQbQ/GFxI/ToDwYfI2IFFgg==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "freebsd" + ] + }, + "node_modules/@yuku-analyzer/binding-linux-arm-gnu": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-linux-arm-gnu/-/binding-linux-arm-gnu-0.6.3.tgz", + "integrity": "sha512-QVMkLA7vqtADSl9sKpX0oDO9X9BmVO6rzlxwU2mJ8vNoYOucOfxOVj1NLW4I3p6m4TW2lIX6zW7kT81HVRmQtw==", + "cpu": [ + "arm" + ], + "libc": [ + "glibc" + ], + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@yuku-analyzer/binding-linux-arm-musl": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-linux-arm-musl/-/binding-linux-arm-musl-0.6.3.tgz", + "integrity": "sha512-MdgimxnvfC4uAMDs0UsQW8wGWi6im+cptlcdQi3l+FS5ouf0tmdIo6O5HoteM6zazUtc+vBEr9g1A4J5eFFKNw==", + "cpu": [ + "arm" + ], + "libc": [ + "musl" + ], + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@yuku-analyzer/binding-linux-arm64-gnu": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-0.6.3.tgz", + "integrity": "sha512-dRYQU8024UvbDnfU3yNDl4NAjLptjkog+Fbd7TsFvQKc9P7rMocC0CLWvsbp6GVu6mo1yATb7GQuSA0V/MBk7g==", + "cpu": [ + "arm64" + ], + "libc": [ + "glibc" + ], + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@yuku-analyzer/binding-linux-arm64-musl": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-linux-arm64-musl/-/binding-linux-arm64-musl-0.6.3.tgz", + "integrity": "sha512-RNj/MBlBYVamdO+Zexxj+tYQiRBPHUMHOLNpCXJ2sraVvKc+aT+HzgWwanG1TDL7RR1hfaDxpsmJpLgtDw65nQ==", + "cpu": [ + "arm64" + ], + "libc": [ + "musl" + ], + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@yuku-analyzer/binding-linux-x64-gnu": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-linux-x64-gnu/-/binding-linux-x64-gnu-0.6.3.tgz", + "integrity": "sha512-gUHi0GkcJOfGc+RHkqyTSpjyLRNIm0cZUSEOVdH309iXDEMqD9E0Fz3kUoxepIrBwkHDAjF3xYrp0hEGogfkUw==", + "cpu": [ + "x64" + ], + "libc": [ + "glibc" + ], + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@yuku-analyzer/binding-linux-x64-musl": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-linux-x64-musl/-/binding-linux-x64-musl-0.6.3.tgz", + "integrity": "sha512-7xIqcdYwyf6mSCJid9C0ZMxd6KdN3S72Ywnws55Wz3O7XWvVscZj+51DHAeqKklsD0gXey4evLQkaDtKxd8jBQ==", + "cpu": [ + "x64" + ], + "libc": [ + "musl" + ], + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@yuku-analyzer/binding-win32-arm64": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-win32-arm64/-/binding-win32-arm64-0.6.3.tgz", + "integrity": "sha512-tyU9RPF0reQ4Lu2JKDhsSZZIqSHJZdO3QD7vfcJQDhZAU2JBvfubpVejvf3uvAS+/2c0Ajfgj/K1JpDyAqVbQA==", + "cpu": [ + "arm64" + ], + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@yuku-analyzer/binding-win32-x64": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@yuku-analyzer/binding-win32-x64/-/binding-win32-x64-0.6.3.tgz", + "integrity": "sha512-84vgw5+SNDYhTJYFkLkgobYM++1IbN8fPY4QIRVSBuLZftHvuaKQnrjQTQkXvmPc0fiebY8ITnxNjHydDrItDA==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@yuku-toolchain/types": { + "version": "0.5.43", + "resolved": "https://registry.npmjs.org/@yuku-toolchain/types/-/types-0.5.43.tgz", + "integrity": "sha512-kSpvPntnXw5+lYjO71ffBEnQ5ycQ74KGIYknh0TS4xeyCuBkOqxyJumxZkMhLBBUCLjDAbx2+Icnr3Zh4ftjpQ==", + "license": "MIT" + }, "node_modules/a-sync-waterfall": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/a-sync-waterfall/-/a-sync-waterfall-1.0.1.tgz", @@ -15951,67 +15888,6 @@ "node": ">=20" } }, - "node_modules/cross-fetch": { - "version": "4.1.0", - "resolved": "https://registry.npmjs.org/cross-fetch/-/cross-fetch-4.1.0.tgz", - "integrity": "sha512-uKm5PU+MHTootlWEY+mZ4vvXoCn4fLQxT9dSc1sXVMSFkINTJVN8cAQROpwcKm8bJ/c7rgZVIBWzH5T78sNZZw==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "node-fetch": "^2.7.0" - } - }, - "node_modules/cross-fetch/node_modules/node-fetch": { - "version": "2.7.0", - "resolved": "https://registry.npmjs.org/node-fetch/-/node-fetch-2.7.0.tgz", - "integrity": "sha512-c4FRfUm/dbcWZ7U+1Wq0AwCyFL+3nt2bEw05wfxSz+DWpWsitgmSgYmy2dQdWyKC1694ELPqMs/YzUSNozLt8A==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "whatwg-url": "^5.0.0" - }, - "engines": { - "node": "4.x || >=6.0.0" - }, - "peerDependencies": { - "encoding": "^0.1.0" - }, - "peerDependenciesMeta": { - "encoding": { - "optional": true - } - } - }, - "node_modules/cross-fetch/node_modules/tr46": { - "version": "0.0.3", - "resolved": "https://registry.npmjs.org/tr46/-/tr46-0.0.3.tgz", - "integrity": "sha512-N3WMsuqV66lT30CrXNbEjx4GEwlow3v6rr4mCcv6prnfwhS01rkgyFdjPNBYd9br7LpXV1+Emh01fHnq2Gdgrw==", - "dev": true, - "license": "MIT", - "optional": true - }, - "node_modules/cross-fetch/node_modules/webidl-conversions": { - "version": "3.0.1", - "resolved": "https://registry.npmjs.org/webidl-conversions/-/webidl-conversions-3.0.1.tgz", - "integrity": "sha512-2JAn3z8AR6rjK8Sm8orRC0h/bcl/DqL7tRPdGZ4I1CjdF+EaMLmYxBHyXuKL849eucPFhvBoxMsflfOb8kxaeQ==", - "dev": true, - "license": "BSD-2-Clause", - "optional": true - }, - "node_modules/cross-fetch/node_modules/whatwg-url": { - "version": "5.0.0", - "resolved": "https://registry.npmjs.org/whatwg-url/-/whatwg-url-5.0.0.tgz", - "integrity": "sha512-saE57nupxk6v3HY35+jzBwYa0rKSy0XR8JSxZPwgLr7ys0IBzhGviA1/TUGJLmSVqs8pb9AnvICXEuOHLprYTw==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "tr46": "~0.0.3", - "webidl-conversions": "^3.0.0" - } - }, "node_modules/cross-spawn": { "version": "7.0.6", "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz", @@ -17953,9 +17829,9 @@ } }, "node_modules/es-toolkit": { - "version": "1.45.1", - "resolved": "https://registry.npmjs.org/es-toolkit/-/es-toolkit-1.45.1.tgz", - "integrity": "sha512-/jhoOj/Fx+A+IIyDNOvO3TItGmlMKhtX8ISAHKE90c4b/k1tqaqEZ+uUqfpU8DMnW5cgNJv606zS55jGvza0Xw==", + "version": "1.49.0", + "resolved": "https://registry.npmjs.org/es-toolkit/-/es-toolkit-1.49.0.tgz", + "integrity": "sha512-G5iZ6Pc/FNRY/soKZHC+TxGDD83rHUDXxzaWhGCX44vAv/tMs56WMusnm/KMNK+luUPsgA9U28cGr4RDlSzL2g==", "license": "MIT", "workspaces": [ "docs", @@ -18133,9 +18009,9 @@ } }, "node_modules/eslint": { - "version": "9.39.4", - "resolved": "https://registry.npmjs.org/eslint/-/eslint-9.39.4.tgz", - "integrity": "sha512-XoMjdBOwe/esVgEvLmNsD3IRHkm7fbKIUGvrleloJXUZgDHig2IPWNniv+GwjyJXzuNqVjlr5+4yVUZjycJwfQ==", + "version": "9.39.5", + "resolved": "https://registry.npmjs.org/eslint/-/eslint-9.39.5.tgz", + "integrity": "sha512-DgZS62aPLXKlnxILS/AYCoRvHaZeXceIzlXPkkGGzJWSow1aEk0lbTlxUSlyjC8jcaKxAdOnTDz+o1JFSBsyjw==", "dev": true, "license": "MIT", "dependencies": { @@ -18144,8 +18020,8 @@ "@eslint/config-array": "^0.21.2", "@eslint/config-helpers": "^0.4.2", "@eslint/core": "^0.17.0", - "@eslint/eslintrc": "^3.3.5", - "@eslint/js": "9.39.4", + "@eslint/eslintrc": "^3.3.6", + "@eslint/js": "9.39.5", "@eslint/plugin-kit": "^0.4.1", "@humanfs/node": "^0.16.6", "@humanwhocodes/module-importer": "^1.0.1", @@ -18486,9 +18362,9 @@ } }, "node_modules/eslint-plugin-sonarjs": { - "version": "4.1.0", - "resolved": "https://registry.npmjs.org/eslint-plugin-sonarjs/-/eslint-plugin-sonarjs-4.1.0.tgz", - "integrity": "sha512-rh+FlVz0yfd2RNIb6WqSkuGh0addX/Qi5scwQ5FphXDFrM6fZKcxP1+attJ78yUKcyYfiu6MTaISPpAFPzqRJw==", + "version": "4.2.0", + "resolved": "https://registry.npmjs.org/eslint-plugin-sonarjs/-/eslint-plugin-sonarjs-4.2.0.tgz", + "integrity": "sha512-bqADfuNtTL7VK6RU29eoiFTtaaBKIpVPuX3bOl+rBpWSBa0zIBVZlqZNZQjfP6s4iXkAJokv5IsD8OsACkwApg==", "dev": true, "license": "LGPL-3.0-only", "dependencies": { @@ -18496,14 +18372,14 @@ "builtin-modules": "^3.3.0", "bytes": "^3.1.2", "functional-red-black-tree": "^1.0.1", - "globals": "^17.6.0", + "globals": "^17.7.0", "jsx-ast-utils-x": "^0.1.0", "lodash.merge": "^4.6.2", "minimatch": "^10.2.5", "scslre": "^0.3.0", - "semver": "^7.8.4", + "semver": "^7.8.5", "ts-api-utils": "^2.5.0", - "typescript": ">=5", + "typescript": ">=5 <6.1.0", "yaml": "^2.9.0" }, "peerDependencies": { @@ -18534,9 +18410,9 @@ } }, "node_modules/eslint-plugin-sonarjs/node_modules/globals": { - "version": "17.6.0", - "resolved": "https://registry.npmjs.org/globals/-/globals-17.6.0.tgz", - "integrity": "sha512-sepffkT8stwnIYbsMBpoCHJuJM5l98FUF2AnE07hfvE0m/qp3R586hw4jF4uadbhvg1ooIdzuu7CsfD2jzCaNA==", + "version": "17.7.0", + "resolved": "https://registry.npmjs.org/globals/-/globals-17.7.0.tgz", + "integrity": "sha512-Czmyns5dUsq4seFBR/Kdydhmo8y9kC79hiSkPn0YcGtNnYWnrgt0vjrSjx9tspoDGWm2CMarffRuLjM4xUz8xg==", "dev": true, "license": "MIT", "engines": { @@ -18563,9 +18439,9 @@ } }, "node_modules/eslint-plugin-sonarjs/node_modules/semver": { - "version": "7.8.4", - "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.4.tgz", - "integrity": "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA==", + "version": "7.8.5", + "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", + "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", "dev": true, "license": "ISC", "bin": { @@ -19051,9 +18927,9 @@ "license": "MIT" }, "node_modules/fast-check": { - "version": "4.8.0", - "resolved": "https://registry.npmjs.org/fast-check/-/fast-check-4.8.0.tgz", - "integrity": "sha512-GOJ158CUMnN6cSahsv4+ExARvIDuzzinFjkp0E9WtiBa5zcVeLozVkWaE4IzFcc+Y48Wp1EDlUZsXRyAztQcSg==", + "version": "4.9.0", + "resolved": "https://registry.npmjs.org/fast-check/-/fast-check-4.9.0.tgz", + "integrity": "sha512-7ms6T7SybUev/PQITciI0yLM2pOSFy5zpG8Ty7tQofcVaQUvrMXp6CBwqF6fThLCLOrfBtuHAtwq6Yu4XPCllg==", "dev": true, "funding": [ { @@ -19318,14 +19194,6 @@ "node": "^12.20 || >= 14.13" } }, - "node_modules/fetch-retry": { - "version": "5.0.6", - "resolved": "https://registry.npmjs.org/fetch-retry/-/fetch-retry-5.0.6.tgz", - "integrity": "sha512-3yurQZ2hD9VISAhJJP9bpYFNQrHHBXE2JxxjY5aLEcDi46RmAzJE2OC9FAde0yis5ElW0jTTzs0zfg/Cca4XqQ==", - "dev": true, - "license": "MIT", - "optional": true - }, "node_modules/fetch-socks": { "version": "1.3.3", "resolved": "https://registry.npmjs.org/fetch-socks/-/fetch-socks-1.3.3.tgz", @@ -19727,9 +19595,9 @@ } }, "node_modules/fumadocs-core": { - "version": "16.11.1", - "resolved": "https://registry.npmjs.org/fumadocs-core/-/fumadocs-core-16.11.1.tgz", - "integrity": "sha512-tKuh1AKoVTb+f7IoAOM2cfz5djd3YhePeqA95q6mf422gEvDTeJms23OJ+icYRWZ6ryNQ5W/ZsgKEe87M5HVYg==", + "version": "16.11.5", + "resolved": "https://registry.npmjs.org/fumadocs-core/-/fumadocs-core-16.11.5.tgz", + "integrity": "sha512-YrHjS09+QYYKOSTGyiZbxF/VDs7ciMcjurYBGfmYqtzdj14k7Ho0HX9c6VuvG54YsYHQs5mGWemT1TXm7vDBaA==", "license": "MIT", "dependencies": { "@orama/orama": "^3.1.18", @@ -19737,7 +19605,6 @@ "github-slugger": "^2.0.0", "hast-util-to-estree": "^3.1.3", "hast-util-to-jsx-runtime": "^2.3.6", - "js-yaml": "^5.2.1", "mdast-util-mdx": "^3.0.0", "mdast-util-to-markdown": "^2.1.2", "remark": "^15.0.1", @@ -19748,7 +19615,8 @@ "tinyglobby": "^0.2.17", "unified": "^11.0.5", "unist-util-visit": "^5.1.0", - "vfile": "^6.0.3" + "vfile": "^6.0.3", + "yaml": "^2.9.0" }, "peerDependencies": { "@mdx-js/mdx": "*", @@ -19828,9 +19696,9 @@ } }, "node_modules/fumadocs-mdx": { - "version": "15.1.0", - "resolved": "https://registry.npmjs.org/fumadocs-mdx/-/fumadocs-mdx-15.1.0.tgz", - "integrity": "sha512-2nDusSlYFuNVcyB51jgY3tA3r01ALTwoURrMDNoc7cbJKZ2sac/PW+CDq6SHTArkgRMmFiKYQGfspJdjgTtPTg==", + "version": "15.2.0", + "resolved": "https://registry.npmjs.org/fumadocs-mdx/-/fumadocs-mdx-15.2.0.tgz", + "integrity": "sha512-+yBP8QYw5wA9LF5eVdMhwbP7KT1OF4B/YfC6PZoD2jz0amZi1B+6QHTI6XoRRSTmhWrI4cL5LU1DspW0itk+NA==", "license": "MIT", "dependencies": { "@mdx-js/mdx": "^3.1.1", @@ -19839,7 +19707,7 @@ "esbuild": "^0.28.1", "estree-util-value-to-estree": "^3.5.0", "github-slugger": "^2.0.0", - "js-yaml": "^5.2.1", + "magic-string": "^0.30.21", "mdast-util-mdx": "^3.0.0", "picocolors": "^1.1.1", "picomatch": "^4.0.5", @@ -19849,6 +19717,8 @@ "unist-util-remove-position": "^5.0.0", "unist-util-visit": "^5.1.0", "vfile": "^6.0.3", + "yaml": "^2.9.0", + "yuku-analyzer": "^0.6.3", "zod": "^4.4.3" }, "bin": { @@ -19901,26 +19771,26 @@ } }, "node_modules/fumadocs-ui": { - "version": "16.11.1", - "resolved": "https://registry.npmjs.org/fumadocs-ui/-/fumadocs-ui-16.11.1.tgz", - "integrity": "sha512-Dq819PFV4RGhAI9Wd4erSCiRlEDLVOZae+kgE5LeOKFH8mbKX49U8N17ldFOhdkC9EZpxMZdEKul77RDgFHQww==", + "version": "16.11.5", + "resolved": "https://registry.npmjs.org/fumadocs-ui/-/fumadocs-ui-16.11.5.tgz", + "integrity": "sha512-Eda7x2Hk7E1iIjZ4uES0xxGr25Z72efRM5kP8sbgLSLhWg8TDCyWddvKAkzXIq8bupPOuJkdZa/YVvXbCktIEA==", "license": "MIT", "dependencies": { "@fuma-translate/react": "^1.0.2", - "@fumadocs/tailwind": "0.1.0", - "@radix-ui/react-accordion": "^1.2.15", - "@radix-ui/react-collapsible": "^1.1.15", - "@radix-ui/react-dialog": "^1.1.18", + "@fumadocs/tailwind": "0.1.1", + "@radix-ui/react-accordion": "^1.2.16", + "@radix-ui/react-collapsible": "^1.1.16", + "@radix-ui/react-dialog": "^1.1.19", "@radix-ui/react-direction": "^1.1.2", - "@radix-ui/react-navigation-menu": "^1.2.17", - "@radix-ui/react-popover": "^1.1.18", - "@radix-ui/react-presence": "^1.1.6", - "@radix-ui/react-scroll-area": "^1.2.13", + "@radix-ui/react-navigation-menu": "^1.2.18", + "@radix-ui/react-popover": "^1.1.19", + "@radix-ui/react-presence": "^1.1.7", + "@radix-ui/react-scroll-area": "^1.2.14", "@radix-ui/react-slot": "^1.3.0", - "@radix-ui/react-tabs": "^1.1.16", + "@radix-ui/react-tabs": "^1.1.17", "class-variance-authority": "^0.7.1", "cnfast": "^0.0.8", - "lucide-react": "^1.23.0", + "lucide-react": "^1.24.0", "motion": "^12.42.2", "next-themes": "^0.4.6", "react-remove-scroll": "^2.7.2", @@ -19930,18 +19800,15 @@ "unist-util-visit": "^5.1.0" }, "peerDependencies": { - "@takumi-rs/image-response": "*", "@types/mdx": "*", "@types/react": "*", - "fumadocs-core": "16.11.1", + "fumadocs-core": "16.11.5", "next": "16.x.x", "react": "^19.2.0", - "react-dom": "^19.2.0" + "react-dom": "^19.2.0", + "takumi-js": "*" }, "peerDependenciesMeta": { - "@takumi-rs/image-response": { - "optional": true - }, "@types/mdx": { "optional": true }, @@ -19950,6 +19817,9 @@ }, "next": { "optional": true + }, + "takumi-js": { + "optional": true } } }, @@ -21267,14 +21137,6 @@ "node": "^22.15.0 || ^24.0.0 || >=26.0.0" } }, - "node_modules/http-status-codes": { - "version": "2.3.0", - "resolved": "https://registry.npmjs.org/http-status-codes/-/http-status-codes-2.3.0.tgz", - "integrity": "sha512-RJ8XvFvpPM/Dmc5SV+dC4y5PCeOhT3x1Hq0NU3rjGeg5a/CqlhZ7uudknPwZFz4aeAXDcbAyaeP7GAo9lvngtA==", - "dev": true, - "license": "MIT", - "optional": true - }, "node_modules/http-z": { "version": "8.1.1", "resolved": "https://registry.npmjs.org/http-z/-/http-z-8.1.1.tgz", @@ -21753,9 +21615,9 @@ } }, "node_modules/icu-minify": { - "version": "4.13.1", - "resolved": "https://registry.npmjs.org/icu-minify/-/icu-minify-4.13.1.tgz", - "integrity": "sha512-nFYW2im0WJ3RUVZwabd71J8QTZRtkK1xxZBY7klg7a6KS/os17LZSj9q1VhbRSnk3S8Mv2I7F1izr/aEJ6cUsw==", + "version": "4.13.2", + "resolved": "https://registry.npmjs.org/icu-minify/-/icu-minify-4.13.2.tgz", + "integrity": "sha512-XhYQTEnBXBCyF6ERiwItFweoOXDUciujaGjIWCA7RhOCEPDfhVSTtuwfRq6HdVk7tKnYJh+yaurP96zGnaKsPg==", "funding": [ { "type": "individual", @@ -22669,19 +22531,19 @@ } }, "node_modules/intl-messageformat": { - "version": "11.2.9", - "resolved": "https://registry.npmjs.org/intl-messageformat/-/intl-messageformat-11.2.9.tgz", - "integrity": "sha512-cGzymZerpDhVXRKjKLgXKda9gI29TU2o88L7gwNMHp3WZVxA/0c5tX52udXbW9JklDApolvMXZG6Dhhdz5eirA==", + "version": "11.2.11", + "resolved": "https://registry.npmjs.org/intl-messageformat/-/intl-messageformat-11.2.11.tgz", + "integrity": "sha512-aDG5bvFRbQvRoT2Bh9FV6yV8t7o0MjEGknZ6pnin5Wt52PJwaBOHDfvz+oPEe78Pl3InQYKugBgCdXijLj6viQ==", "license": "BSD-3-Clause", "dependencies": { - "@formatjs/fast-memoize": "3.1.6", - "@formatjs/icu-messageformat-parser": "3.5.12" + "@formatjs/fast-memoize": "3.1.7", + "@formatjs/icu-messageformat-parser": "3.5.14" } }, "node_modules/intl-messageformat/node_modules/@formatjs/fast-memoize": { - "version": "3.1.6", - "resolved": "https://registry.npmjs.org/@formatjs/fast-memoize/-/fast-memoize-3.1.6.tgz", - "integrity": "sha512-H5aexk1Le7T9TPmscacZ+1pR6CTa2n1wq+HDVGXhH8TzUlQQpeXzZs91dRtmFHrbeNbjPFPfQujUqm7MHgVoXQ==", + "version": "3.1.7", + "resolved": "https://registry.npmjs.org/@formatjs/fast-memoize/-/fast-memoize-3.1.7.tgz", + "integrity": "sha512-zXfhLpvA6T7+efdt9JLbBwZ00tT7NsBMDVnDu8rpHeNNv8KfRZAMo2gkG0k9lK/Nzc//3kJ9pImsfuJxk3KhUA==", "license": "MIT" }, "node_modules/ioredis": { @@ -24409,9 +24271,9 @@ "integrity": "sha512-Ls993zuzfayK269Svk9hzpeGUKob/sIgZzyHYdjQoAdQetRKpOLj+k/QQQ/6Qi0Yz65mlROrfd+Ev+1+7dz9Kw==" }, "node_modules/knip": { - "version": "6.25.0", - "resolved": "https://registry.npmjs.org/knip/-/knip-6.25.0.tgz", - "integrity": "sha512-Q3n41VjOOB/aqsbxb8kallAcFKrUz3b2S5fD5pTODljVpP01t+rvAgy2x3j0Cq8yEpRRHNdar1vHuqFfGuIakQ==", + "version": "6.27.0", + "resolved": "https://registry.npmjs.org/knip/-/knip-6.27.0.tgz", + "integrity": "sha512-CngYEYrD0n20N06FXA8n3u/0Wnnugoa+B9k14OP+iKIgkCHuzvIdsP3nfwjhByoc1WfogpxfiriMboAXFETDUw==", "dev": true, "funding": [ { @@ -25817,9 +25679,9 @@ } }, "node_modules/lucide-react": { - "version": "1.23.0", - "resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.23.0.tgz", - "integrity": "sha512-38BpJcD0JhFosxHApP/BYsBetLpQFRoTRzEzstM/XCc3jsAG7wqaY1lgVwxiUe3xqYE+lNxo2PkCmYwXWrwwIw==", + "version": "1.24.0", + "resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.24.0.tgz", + "integrity": "sha512-YT6mBD8lGKkg4nM39enlm94/sfJIiW0YKUT60fBy4YK8tai31ylg1VhGNWxkpSKHo9UagfnZqwIff3HTDQwXeA==", "license": "ISC", "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" @@ -25829,7 +25691,6 @@ "version": "0.30.21", "resolved": "https://registry.npmjs.org/magic-string/-/magic-string-0.30.21.tgz", "integrity": "sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ==", - "dev": true, "license": "MIT", "dependencies": { "@jridgewell/sourcemap-codec": "^1.5.5" @@ -25911,9 +25772,9 @@ } }, "node_modules/marked": { - "version": "18.0.5", - "resolved": "https://registry.npmjs.org/marked/-/marked-18.0.5.tgz", - "integrity": "sha512-S6GcvALHg6K4ohtu4E7x0a1AqhAjp6cV8KhLSyN9qVapnzJkusVBxZRcIU9AeYsbe6P1hKDusSbEOzGyyuce6w==", + "version": "18.0.6", + "resolved": "https://registry.npmjs.org/marked/-/marked-18.0.6.tgz", + "integrity": "sha512-MrV5puXBfuiy6wl6DLaq3BtIJQAJToAd5zt/ZKhRfGRAuFPALE7/4Y7jnxRQoEgK/pBgurGqLyAuRgZ2xOjr6w==", "license": "MIT", "bin": { "marked": "bin/marked.js" @@ -25981,9 +25842,9 @@ } }, "node_modules/material-symbols": { - "version": "0.45.6", - "resolved": "https://registry.npmjs.org/material-symbols/-/material-symbols-0.45.6.tgz", - "integrity": "sha512-sPsLRMFIRETZKVrkwc5VW8lH6FT62vMuTmmQVjftFQ1UJXhK0dQX8isGDDwR72N+rN3oYATC2w9Q5oZB8lczdw==", + "version": "0.45.8", + "resolved": "https://registry.npmjs.org/material-symbols/-/material-symbols-0.45.8.tgz", + "integrity": "sha512-1kpL3jl+/f6W34fmENPYMyftCmBNXUannoZ8XzlMsuEEFlIxesX+Z6ZCj2qSJGxzqwJ5yE/3x1XzVPsycieG2g==", "license": "Apache-2.0" }, "node_modules/math-intrinsics": { @@ -28097,9 +27958,9 @@ } }, "node_modules/next-intl": { - "version": "4.13.1", - "resolved": "https://registry.npmjs.org/next-intl/-/next-intl-4.13.1.tgz", - "integrity": "sha512-aS8KTA+nNhSNJJBlIhxgvU135WzoObwzFwav4wTDti/Gmhxqe0fs/Q343igo8Z7HGqPB/xgmoagwySZAlHmIfA==", + "version": "4.13.2", + "resolved": "https://registry.npmjs.org/next-intl/-/next-intl-4.13.2.tgz", + "integrity": "sha512-iCYycEP7/PE+1ue4MWBuz1qns6ESA+SGzuXxNMNN1qiYUh2fLhJPiZvHW1zZC7zMTn44aJUglOVZxUb1+hUz6g==", "funding": [ { "type": "individual", @@ -28111,11 +27972,11 @@ "@formatjs/intl-localematcher": "^0.8.1", "@parcel/watcher": "^2.4.1", "@swc/core": "^1.15.2", - "icu-minify": "^4.13.1", + "icu-minify": "^4.13.2", "negotiator": "^1.0.0", - "next-intl-swc-plugin-extractor": "^4.13.1", + "next-intl-swc-plugin-extractor": "^4.13.2", "po-parser": "^2.1.1", - "use-intl": "^4.13.1" + "use-intl": "^4.13.2" }, "peerDependencies": { "next": "^12.0.0 || ^13.0.0 || ^14.0.0 || ^15.0.0 || ^16.0.0", @@ -28128,9 +27989,9 @@ } }, "node_modules/next-intl-swc-plugin-extractor": { - "version": "4.13.1", - "resolved": "https://registry.npmjs.org/next-intl-swc-plugin-extractor/-/next-intl-swc-plugin-extractor-4.13.1.tgz", - "integrity": "sha512-RhlH2DR1ViEXzcX7G3tDXAvzNrBL2Ph54Hq/q/9oP9eXQV/okh3UQpA/lx2k9U5Ck83CZOSgH0eFTNi5U+zyXw==", + "version": "4.13.2", + "resolved": "https://registry.npmjs.org/next-intl-swc-plugin-extractor/-/next-intl-swc-plugin-extractor-4.13.2.tgz", + "integrity": "sha512-O30N/Y4ifzRe5Sz80jD1Qkg4VY6Zfef4SbHNNE166QkHswBf3/Kygpdd1X52sUUwTCk6uvdisO8ybfAN/VYJHQ==", "license": "MIT" }, "node_modules/next-themes": { @@ -28892,9 +28753,9 @@ "license": "MIT" }, "node_modules/omniglyph": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/omniglyph/-/omniglyph-1.0.2.tgz", - "integrity": "sha512-GGLet99n3HVxOx3WuNPda4B0ETptX9SA8h1fnm/AYSXmvsKXE3mN11Ae2jfX8ldAOA/rVJRXHzPhNkXsRHzMsg==", + "version": "1.3.1", + "resolved": "https://registry.npmjs.org/omniglyph/-/omniglyph-1.3.1.tgz", + "integrity": "sha512-6QnZCoXYczjsPN2x+XpbimimjO6kCoSZUzsdSvoKjtw28U1U724VgLICBNaLX4FFs5jd7SrYNNs9Aee2iIkcoA==", "license": "MIT", "dependencies": { "gpt-tokenizer": "^3.4.0" @@ -29093,25 +28954,6 @@ } } }, - "node_modules/openapi-fetch": { - "version": "0.8.2", - "resolved": "https://registry.npmjs.org/openapi-fetch/-/openapi-fetch-0.8.2.tgz", - "integrity": "sha512-4g+NLK8FmQ51RW6zLcCBOVy/lwYmFJiiT+ckYZxJWxUxH4XFhsNcX2eeqVMfVOi+mDNFja6qDXIZAz2c5J/RVw==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "openapi-typescript-helpers": "^0.0.5" - } - }, - "node_modules/openapi-typescript-helpers": { - "version": "0.0.5", - "resolved": "https://registry.npmjs.org/openapi-typescript-helpers/-/openapi-typescript-helpers-0.0.5.tgz", - "integrity": "sha512-MRffg93t0hgGZbYTxg60hkRIK2sRuEOHEtCUgMuLgbCC33TMQ68AmxskzUlauzZYD47+ENeGV/ElI7qnWqrAxA==", - "dev": true, - "license": "MIT", - "optional": true - }, "node_modules/opener": { "version": "1.5.2", "resolved": "https://registry.npmjs.org/opener/-/opener-1.5.2.tgz", @@ -29366,21 +29208,6 @@ "url": "https://github.com/sponsors/sindresorhus" } }, - "node_modules/p-queue-compat": { - "version": "1.0.225", - "resolved": "https://registry.npmjs.org/p-queue-compat/-/p-queue-compat-1.0.225.tgz", - "integrity": "sha512-SdfGSQSJJpD7ZR+dJEjjn9GuuBizHPLW/yarJpXnmrHRruzrq7YM8OqsikSrKeoPv+Pi1YXw9IIBSIg5WveQHA==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "eventemitter3": "5.x", - "p-timeout-compat": "^1.0.3" - }, - "engines": { - "node": ">=12" - } - }, "node_modules/p-queue/node_modules/eventemitter3": { "version": "4.0.7", "resolved": "https://registry.npmjs.org/eventemitter3/-/eventemitter3-4.0.7.tgz", @@ -29429,17 +29256,6 @@ "node": ">=8" } }, - "node_modules/p-timeout-compat": { - "version": "1.0.8", - "resolved": "https://registry.npmjs.org/p-timeout-compat/-/p-timeout-compat-1.0.8.tgz", - "integrity": "sha512-+7LpKr1ilnWU0LbV2r+Wz4srwMcFTUysmgL824ZxJcZP3u4Hyi/D/39pbyEs4j0XXCHvbv069+LDPxlCijfVRQ==", - "dev": true, - "license": "MIT", - "optional": true, - "engines": { - "node": ">=12" - } - }, "node_modules/pac-proxy-agent": { "version": "9.1.0", "resolved": "https://registry.npmjs.org/pac-proxy-agent/-/pac-proxy-agent-9.1.0.tgz", @@ -30393,9 +30209,9 @@ } }, "node_modules/prettier": { - "version": "3.9.4", - "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.9.4.tgz", - "integrity": "sha512-yWG/o/4oJfo036EKAfK6ACAoDOfHeRHx4tuxkfBZiauURiaSmYwlpOr5LQqKtIkRD2z1PLteme2WoxEnj4tHTg==", + "version": "3.9.5", + "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.9.5.tgz", + "integrity": "sha512-/FVl766LpUfB5vXgCYOYa0MeV/441Ia99AeICQIQFTY/Nw0roZwULcXpku5i1/m5kt/baz+s4Zogspd839HSMg==", "dev": true, "license": "MIT", "bin": { @@ -30540,9 +30356,9 @@ } }, "node_modules/promptfoo": { - "version": "0.121.18", - "resolved": "https://registry.npmjs.org/promptfoo/-/promptfoo-0.121.18.tgz", - "integrity": "sha512-avytaJ3Vi043Cp/LHRNstKK7PzaDso5QvPa1llMAsISfG8uC7w3mKATGlLcO8Qo6SIhmibfz2JUQvG0EuFuJ0g==", + "version": "0.121.19", + "resolved": "https://registry.npmjs.org/promptfoo/-/promptfoo-0.121.19.tgz", + "integrity": "sha512-5YebsCED/bmR9JktH9YNU62Tr1m3ncFMlM2tKrguI8vFFUfvqxhNzUBa3Z6huG7OvDKbi69UpamU4CLtYLDezQ==", "dev": true, "license": "MIT", "workspaces": [ @@ -30550,7 +30366,7 @@ "site" ], "dependencies": { - "@anthropic-ai/sdk": "0.106.0", + "@anthropic-ai/sdk": "0.110.0", "@apidevtools/json-schema-ref-parser": "^15.3.1", "@inquirer/checkbox": "^5.1.0", "@inquirer/confirm": "^6.0.8", @@ -30561,8 +30377,8 @@ "@inquirer/select": "^5.1.0", "@libsql/client": "^0.17.3", "@opentelemetry/api": "^1.9.0", - "@opentelemetry/core": "2.8.0", - "@opentelemetry/exporter-trace-otlp-http": "^0.219.0", + "@opentelemetry/core": "2.9.0", + "@opentelemetry/exporter-trace-otlp-http": "^0.220.0", "@opentelemetry/resources": "^2.6.0", "@opentelemetry/sdk-trace-base": "^2.6.0", "@opentelemetry/sdk-trace-node": "^2.6.0", @@ -30599,7 +30415,7 @@ "http-z": "^8.1.1", "istextorbinary": "^9.5.0", "js-rouge": "^3.2.0", - "js-yaml": "5.2.0", + "js-yaml": "5.2.1", "json5": "^2.2.3", "keyv": "^5.6.0", "keyv-file": "^5.3.3", @@ -30638,7 +30454,7 @@ "node": "^20.20.0 || >=22.22.0" }, "optionalDependencies": { - "@anthropic-ai/claude-agent-sdk": "0.3.195", + "@anthropic-ai/claude-agent-sdk": "0.3.201", "@aws-sdk/client-bedrock-agent-runtime": "^3.1045.0", "@aws-sdk/client-bedrock-runtime": "^3.1045.0", "@aws-sdk/client-s3": "^3.1003.0", @@ -30653,10 +30469,9 @@ "@googleapis/sheets": "^13.0.1", "@huggingface/transformers": "^4.0.0", "@ibm-cloud/watsonx-ai": "^1.7.14", - "@ibm-generative-ai/node-sdk": "^3.2.4", "@modelcontextprotocol/sdk": "^1.29.0", "@openai/agents": "^0.11.3", - "@openai/codex-sdk": "^0.142.3", + "@openai/codex-sdk": "^0.144.0", "@opencode-ai/sdk": "^1.14.33", "@playwright/browser-chromium": "^1.60.0", "@rollup/rollup-linux-x64-gnu": "^4.62.0", @@ -30680,7 +30495,7 @@ "playwright": "^1.60.0", "playwright-extra": "^4.3.6", "read-excel-file": "^9.0.0", - "sharp": "^0.35.1" + "sharp": "^0.35.3" } }, "node_modules/promptfoo/node_modules/@huggingface/jinja": { @@ -30881,29 +30696,6 @@ "url": "https://github.com/sponsors/sindresorhus" } }, - "node_modules/promptfoo/node_modules/js-yaml": { - "version": "5.2.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.2.0.tgz", - "integrity": "sha512-YeLUMlvR4Ou1B119LIaM0r65JvbOBooJDc9yEu0dClb/uSC5P4FrLU8OCCz/HXWvtPoIrR0dRzABTjo1sTN9Bw==", - "dev": true, - "funding": [ - { - "type": "github", - "url": "https://github.com/sponsors/puzrin" - }, - { - "type": "github", - "url": "https://github.com/sponsors/nodeca" - } - ], - "license": "MIT", - "dependencies": { - "argparse": "^2.0.1" - }, - "bin": { - "js-yaml": "bin/js-yaml.mjs" - } - }, "node_modules/promptfoo/node_modules/keyv": { "version": "5.6.0", "resolved": "https://registry.npmjs.org/keyv/-/keyv-5.6.0.tgz", @@ -33988,7 +33780,6 @@ "version": "1.6.1", "resolved": "https://registry.npmjs.org/smol-toml/-/smol-toml-1.6.1.tgz", "integrity": "sha512-dWUG8F5sIIARXih1DTaQAX4SsiTXhInKf1buxdY9DIg4ZYPZK5nGM1VRIYmEbDbsHt7USo99xSLFu5Q1IqTmsg==", - "dev": true, "license": "BSD-3-Clause", "engines": { "node": ">= 18" @@ -35622,9 +35413,9 @@ "license": "0BSD" }, "node_modules/tsx": { - "version": "4.23.0", - "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.0.tgz", - "integrity": "sha512-eUdUIaCr963q2h5u3+QwvYp0+eqPvn+egeqZUm0hwERCqqx1E3kK5ehbGCvqSE5MQAULr67ww0cA3jKc3YkM1w==", + "version": "4.23.1", + "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.1.tgz", + "integrity": "sha512-GQHnkIfxyx1wYCOS/wonik5MVRZU9hi1TEZmzGZSCJB1y9YgoZ8H6itNE/u4suE+yLmOzuE4E5S4TZ/ZX2wcWQ==", "license": "MIT", "dependencies": { "esbuild": "~0.28.0" @@ -35956,16 +35747,16 @@ } }, "node_modules/typescript-eslint": { - "version": "8.63.0", - "resolved": "https://registry.npmjs.org/typescript-eslint/-/typescript-eslint-8.63.0.tgz", - "integrity": "sha512-xgwXyzG4sK9ALkBxbyGkTMMOS+imnW65iPhxCQMK83KhxyoDNW7l+IDqEf9vMdoUidHpOoS967RCq4eMiTexwQ==", + "version": "8.64.0", + "resolved": "https://registry.npmjs.org/typescript-eslint/-/typescript-eslint-8.64.0.tgz", + "integrity": "sha512-0qg+pDNMnqYzqH9AnNK+39tejHvsShUOUUoRUgtnTGE7QuMZhiFDnozq8nHJVq+Wae6NMLKNWLg5WmkcC/ndyQ==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/eslint-plugin": "8.63.0", - "@typescript-eslint/parser": "8.63.0", - "@typescript-eslint/typescript-estree": "8.63.0", - "@typescript-eslint/utils": "8.63.0" + "@typescript-eslint/eslint-plugin": "8.64.0", + "@typescript-eslint/parser": "8.64.0", + "@typescript-eslint/typescript-estree": "8.64.0", + "@typescript-eslint/utils": "8.64.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -36415,9 +36206,9 @@ } }, "node_modules/use-intl": { - "version": "4.13.1", - "resolved": "https://registry.npmjs.org/use-intl/-/use-intl-4.13.1.tgz", - "integrity": "sha512-UU5C3zAC7yVg3m7rq5C8VF5J5jhAfvS19Wi9bPNCB9xB7jQYBsUcrqfdxs4Mxl9XR3x6BDB5K++iAw7/rcm3gg==", + "version": "4.13.2", + "resolved": "https://registry.npmjs.org/use-intl/-/use-intl-4.13.2.tgz", + "integrity": "sha512-p6/gCromeBoec+wEuOIkPaytH77RBjW94KWv8MRsNrbI4UOx8QgwNhgl9lbPOKM/b0dpanLhCYztLEH5yQUjtg==", "funding": [ { "type": "individual", @@ -36428,7 +36219,7 @@ "dependencies": { "@formatjs/fast-memoize": "^3.1.0", "@schummar/icu-type-parser": "1.21.5", - "icu-minify": "^4.13.1", + "icu-minify": "^4.13.2", "intl-messageformat": "^11.1.0" }, "peerDependencies": { @@ -37357,9 +37148,9 @@ } }, "node_modules/ws": { - "version": "8.21.0", - "resolved": "https://registry.npmjs.org/ws/-/ws-8.21.0.tgz", - "integrity": "sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g==", + "version": "8.21.1", + "resolved": "https://registry.npmjs.org/ws/-/ws-8.21.1.tgz", + "integrity": "sha512-+0NTnW77fFN/DjQi6k/Sq/Yvk4Sgajw7urW8V+asjXnRgDs9gyGkdb7EzgfhA4goXsRIZKE28fzIXBHEzhuiWw==", "license": "MIT", "engines": { "node": ">=10.0.0" @@ -37580,7 +37371,6 @@ "version": "2.9.0", "resolved": "https://registry.npmjs.org/yaml/-/yaml-2.9.0.tgz", "integrity": "sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA==", - "dev": true, "license": "ISC", "bin": { "yaml": "bin.mjs" @@ -37733,6 +37523,28 @@ "integrity": "sha512-0LPOt3AxKqMdFBZA3HBAt/t/8vIKq7VaQYbuA8WxCgung+p9TVyKRYdpvCb80HcdTN2NkbIKbhNwKUfm3tQywQ==", "license": "MIT" }, + "node_modules/yuku-analyzer": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/yuku-analyzer/-/yuku-analyzer-0.6.3.tgz", + "integrity": "sha512-RQ02dPtOa5d2AA3Np45EWD3EJUwZDruCrMMulPwUT/9GK1P7aKAhjaxG4Jv/1qqzMUIt0RUe3Dn2RR6d7+qTrA==", + "license": "MIT", + "dependencies": { + "@yuku-toolchain/types": "0.5.43" + }, + "optionalDependencies": { + "@yuku-analyzer/binding-darwin-arm64": "0.6.3", + "@yuku-analyzer/binding-darwin-x64": "0.6.3", + "@yuku-analyzer/binding-freebsd-x64": "0.6.3", + "@yuku-analyzer/binding-linux-arm-gnu": "0.6.3", + "@yuku-analyzer/binding-linux-arm-musl": "0.6.3", + "@yuku-analyzer/binding-linux-arm64-gnu": "0.6.3", + "@yuku-analyzer/binding-linux-arm64-musl": "0.6.3", + "@yuku-analyzer/binding-linux-x64-gnu": "0.6.3", + "@yuku-analyzer/binding-linux-x64-musl": "0.6.3", + "@yuku-analyzer/binding-win32-arm64": "0.6.3", + "@yuku-analyzer/binding-win32-x64": "0.6.3" + } + }, "node_modules/zod": { "version": "4.4.3", "resolved": "https://registry.npmjs.org/zod/-/zod-4.4.3.tgz", @@ -37808,7 +37620,8 @@ "version": "3.8.49", "dependencies": { "@toon-format/toon": "^2.3.0", - "safe-regex": "^2.1.1" + "safe-regex": "^2.1.1", + "smol-toml": "1.6.1" } } } diff --git a/package.json b/package.json index 937d70363d2..fa272ed3d6d 100644 --- a/package.json +++ b/package.json @@ -49,7 +49,7 @@ "open-sse" ], "engines": { - "node": ">=22.0.0 <23 || >=24.0.0 <27" + "node": ">=22.22.2 <23 || >=24.0.0 <27" }, "keywords": [ "ai", @@ -79,6 +79,11 @@ "gen:provider-reference": "bun scripts/docs/gen-provider-reference.ts", "bench:compression": "bun scripts/compression/benchmark.ts", "eval:compression": "node --import tsx scripts/compression-eval/index.ts", + "eval:router": "node --import tsx scripts/router-eval/index.ts", + "eval:router:compare": "node --import tsx scripts/router-eval/compare.ts", + "eval:router:patch-compare": "node --import tsx scripts/router-eval/patch-compare.ts", + "eval:router:search": "node --import tsx scripts/router-eval/search.ts", + "eval:router:trends": "node --import tsx scripts/router-eval/trends.ts", "release:sync-changelog-i18n": "node scripts/release/sync-changelog-i18n.mjs", "build": "node scripts/build/build-next-isolated.mjs", "build:secure": "OMNIROUTE_BUILD_PROFILE=minimal node scripts/build/build-next-isolated.mjs", @@ -118,6 +123,7 @@ "check:docs-counts": "node scripts/check/check-docs-counts-sync.mjs", "check:deprecated-versions": "node scripts/check/check-deprecated-versions.mjs", "check:compression-budget": "bun scripts/check/check-compression-budget.ts", + "check:router-eval": "node --import tsx scripts/check/check-router-eval-regression.ts", "check:doc-links": "node scripts/check/check-doc-links.mjs", "check:fabricated-docs": "node scripts/check/check-fabricated-docs.mjs --strict", "check:docs-all": "npm run check:docs-sync && npm run check:docs-counts && npm run check:env-doc-sync && npm run check:deprecated-versions && npm run check:doc-links && npm run check:fabricated-docs", @@ -137,6 +143,7 @@ "check:openapi-security-tiers": "node scripts/check/check-openapi-security-tiers.mjs", "check:provider-consistency": "bun scripts/check/check-provider-consistency.ts", "check:provider-assets": "node scripts/check/check-provider-assets.mjs", + "check:nvidia-catalog-drift": "node --import tsx/esm scripts/check/check-nvidia-catalog-drift.ts", "check:fetch-targets": "node scripts/check/check-fetch-targets.mjs", "check:openapi-routes": "node scripts/check/check-openapi-routes.mjs", "check:api-docs-refs": "node scripts/check/check-api-docs-refs.mjs", @@ -285,6 +292,7 @@ "recharts": "^3.8.1", "safe-regex": "^2.1.1", "selfsigned": "^5.5.0", + "smol-toml": "1.6.1", "socks": "^2.8.7", "sql.js": "^1.14.1", "sqlite-vec": "^0.1.9", diff --git a/scripts/check/check-nvidia-catalog-drift.ts b/scripts/check/check-nvidia-catalog-drift.ts new file mode 100644 index 00000000000..bbd279e56cd --- /dev/null +++ b/scripts/check/check-nvidia-catalog-drift.ts @@ -0,0 +1,108 @@ +import { FREE_MODEL_BUDGETS } from "../../open-sse/config/freeModelCatalog.data.ts"; +import reviewedLiveIds from "../../open-sse/config/nvidiaHostedModels.snapshot.json" with { type: "json" }; + +const NVIDIA_MODELS_URL = "https://integrate.api.nvidia.com/v1/models"; + +export interface NvidiaCatalogDrift { + liveCount: number; + reviewedLiveCount: number; + documentedFreeCount: number; + newLiveIds: string[]; + removedLiveIds: string[]; + documentedMissingUpstreamIds: string[]; +} + +function normalizeIds(ids: Iterable): Set { + return new Set( + [...ids].filter((id) => typeof id === "string" && id.trim().length > 0).map((id) => id.trim()) + ); +} + +export function computeNvidiaCatalogDrift( + liveIdsInput: Iterable, + reviewedLiveIdsInput: Iterable, + documentedFreeIdsInput: Iterable +): NvidiaCatalogDrift { + const liveIds = normalizeIds(liveIdsInput); + const reviewedIds = normalizeIds(reviewedLiveIdsInput); + const documentedFreeIds = normalizeIds(documentedFreeIdsInput); + return { + liveCount: liveIds.size, + reviewedLiveCount: reviewedIds.size, + documentedFreeCount: documentedFreeIds.size, + newLiveIds: [...liveIds].filter((id) => !reviewedIds.has(id)).sort(), + removedLiveIds: [...reviewedIds].filter((id) => !liveIds.has(id)).sort(), + documentedMissingUpstreamIds: [...documentedFreeIds].filter((id) => !liveIds.has(id)).sort(), + }; +} + +function printIds(label: string, ids: string[]): void { + console.log(`${label} (${ids.length})`); + for (const id of ids) console.log(` - ${id}`); +} + +async function main(): Promise { + const apiKey = process.env.NVIDIA_API_KEY?.trim(); + if (!apiKey) { + console.error("NVIDIA_API_KEY is required to query the live NVIDIA NIM model catalog."); + process.exitCode = 2; + return; + } + + let liveIds: string[] = []; + try { + const response = await fetch(NVIDIA_MODELS_URL, { + headers: { Authorization: `Bearer ${apiKey}`, Accept: "application/json" }, + signal: AbortSignal.timeout(30_000), + }); + if (!response.ok) { + console.error(`NVIDIA model catalog returned HTTP ${response.status}.`); + process.exitCode = 2; + return; + } + + const body = (await response.json()) as { data?: Array<{ id?: unknown }> } | null; + liveIds = (body && Array.isArray(body.data) ? body.data : []) + .map((model) => model?.id) + .filter((id): id is string => typeof id === "string"); + } catch (error) { + console.error( + "Failed to fetch or parse NVIDIA model catalog:", + error instanceof Error ? error.message : error + ); + process.exitCode = 2; + return; + } + const documentedFreeIds = FREE_MODEL_BUDGETS.filter((model) => model.provider === "nvidia").map( + (model) => model.modelId + ); + const drift = computeNvidiaCatalogDrift(liveIds, reviewedLiveIds, documentedFreeIds); + + console.log( + `NVIDIA catalog: ${drift.liveCount} live model(s), ${drift.reviewedLiveCount} reviewed live model(s), ${drift.documentedFreeCount} documented free/trial model(s).` + ); + printIds("New live models requiring metadata review", drift.newLiveIds); + printIds("Reviewed models removed from the live catalog", drift.removedLiveIds); + printIds( + "Documented free models missing from the live catalog", + drift.documentedMissingUpstreamIds + ); + + if ( + drift.newLiveIds.length || + drift.removedLiveIds.length || + drift.documentedMissingUpstreamIds.length + ) { + console.warn( + "Catalog drift requires review. Availability alone does not prove that a model is free; verify NVIDIA's model page before updating FREE_MODEL_BUDGETS." + ); + if (process.argv.includes("--strict")) process.exitCode = 1; + } else { + console.log("No NVIDIA catalog drift detected."); + } +} + +const isDirectExecution = process.argv[1]?.endsWith("check-nvidia-catalog-drift.ts"); +if (isDirectExecution) { + await main(); +} diff --git a/scripts/check/check-router-eval-regression.ts b/scripts/check/check-router-eval-regression.ts new file mode 100644 index 00000000000..46d696d6b8e --- /dev/null +++ b/scripts/check/check-router-eval-regression.ts @@ -0,0 +1,287 @@ +#!/usr/bin/env node +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; + +type Args = { + baseline: string; + candidate: string; + baselinePatch?: string; + candidatePatch?: string; + output: string; + jsonOutput: string; + patchOutput: string; + patchJsonOutput: string; + artifactDir?: string; + runId: string; + maxAiqDrop: string; + maxCostIncrease: string; + maxPatchAiqDrop: string; + maxPatchCostIncrease: string; + maxPatchLatencyIncrease: string; + maxPatchRegressionIncrease: string; +}; + +type GateManifest = { + schemaVersion: 1; + kind: "router-eval-gate-run"; + runId: string; + generatedAt: string; + command: string[]; + thresholds: { + maxAiqDrop: number; + maxCostIncrease: number; + patch?: { + maxAiqDrop: number; + maxCostIncrease: number; + maxLatencyIncrease: number; + maxRegressionIncrease: number; + }; + }; + inputs: { + baseline: string; + candidate: string; + baselinePatch?: string; + candidatePatch?: string; + }; + outputs: { + markdown: string; + json: string; + patchMarkdown?: string; + patchJson?: string; + }; + environment: { + runtime: "bun" | "node"; + platform: NodeJS.Platform; + }; + result: { + status: number; + }; +}; + +const repoRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); +const isBunRuntime = "Bun" in globalThis; + +function runTypeScriptScript(args: string[]) { + return spawnSync(process.execPath, isBunRuntime ? args : ["--import", "tsx", ...args], { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }); +} +const defaultFixtureDir = path.join(repoRoot, "tests/fixtures/router-eval"); +const defaultArtifactDir = path.join(os.tmpdir(), "omniroute-router-eval"); + +function getArgValue(name: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + if (index < 0) return undefined; + const value = process.argv[index + 1]; + return value && !value.startsWith("--") ? value : undefined; +} + +function readArgs(): Args { + const artifactDir = getArgValue("artifact-dir"); + const runId = getArgValue("run-id") ?? new Date().toISOString().replace(/[:.]/g, "-"); + const retainedDir = artifactDir ? path.join(path.resolve(artifactDir), runId) : undefined; + return { + baseline: getArgValue("baseline") ?? path.join(defaultFixtureDir, "baseline.ndjson"), + candidate: getArgValue("candidate") ?? path.join(defaultFixtureDir, "candidate.ndjson"), + baselinePatch: getArgValue("baseline-patch"), + candidatePatch: getArgValue("candidate-patch"), + output: getArgValue("output") ?? path.join(retainedDir ?? defaultArtifactDir, "router-eval.md"), + jsonOutput: + getArgValue("json-output") ?? + path.join(retainedDir ?? defaultArtifactDir, "router-eval.json"), + patchOutput: + getArgValue("patch-output") ?? + path.join(retainedDir ?? defaultArtifactDir, "patch-comparison.md"), + patchJsonOutput: + getArgValue("patch-json-output") ?? + path.join(retainedDir ?? defaultArtifactDir, "patch-comparison.json"), + artifactDir, + runId, + maxAiqDrop: getArgValue("max-aiq-drop") ?? "1", + maxCostIncrease: getArgValue("max-cost-increase") ?? "0.05", + maxPatchAiqDrop: getArgValue("max-patch-aiq-drop") ?? "1", + maxPatchCostIncrease: getArgValue("max-patch-cost-increase") ?? "0.05", + maxPatchLatencyIncrease: getArgValue("max-patch-latency-increase") ?? "0.05", + maxPatchRegressionIncrease: getArgValue("max-patch-regression-increase") ?? "0", + }; +} + +function ensureReadable(filePath: string, label: string): void { + if (!fs.existsSync(filePath)) { + console.error(`[router-eval] ${label} missing: ${filePath}`); + process.exit(2); + } +} + +function writeRetainedRun(args: Args, status: number): void { + if (!args.artifactDir) return; + + const runDir = path.join(path.resolve(args.artifactDir), args.runId); + const inputDir = path.join(runDir, "inputs"); + fs.mkdirSync(inputDir, { recursive: true }); + + const baselineCopy = path.join(inputDir, "baseline.ndjson"); + const candidateCopy = path.join(inputDir, "candidate.ndjson"); + fs.copyFileSync(args.baseline, baselineCopy); + fs.copyFileSync(args.candidate, candidateCopy); + const baselinePatchCopy = args.baselinePatch + ? path.join(inputDir, "baseline.patch.json") + : undefined; + const candidatePatchCopy = args.candidatePatch + ? path.join(inputDir, "candidate.patch.json") + : undefined; + if (args.baselinePatch && baselinePatchCopy) + fs.copyFileSync(args.baselinePatch, baselinePatchCopy); + if (args.candidatePatch && candidatePatchCopy) + fs.copyFileSync(args.candidatePatch, candidatePatchCopy); + + const manifest: GateManifest = { + schemaVersion: 1, + kind: "router-eval-gate-run", + runId: args.runId, + generatedAt: new Date().toISOString(), + command: process.argv.slice(1), + thresholds: { + maxAiqDrop: Number.parseFloat(args.maxAiqDrop), + maxCostIncrease: Number.parseFloat(args.maxCostIncrease), + ...(args.baselinePatch && args.candidatePatch + ? { + patch: { + maxAiqDrop: Number.parseFloat(args.maxPatchAiqDrop), + maxCostIncrease: Number.parseFloat(args.maxPatchCostIncrease), + maxLatencyIncrease: Number.parseFloat(args.maxPatchLatencyIncrease), + maxRegressionIncrease: Number.parseFloat(args.maxPatchRegressionIncrease), + }, + } + : {}), + }, + inputs: { + baseline: path.relative(runDir, baselineCopy), + candidate: path.relative(runDir, candidateCopy), + ...(baselinePatchCopy && candidatePatchCopy + ? { + baselinePatch: path.relative(runDir, baselinePatchCopy), + candidatePatch: path.relative(runDir, candidatePatchCopy), + } + : {}), + }, + outputs: { + markdown: path.relative(runDir, args.output), + json: path.relative(runDir, args.jsonOutput), + ...(args.baselinePatch && args.candidatePatch + ? { + patchMarkdown: path.relative(runDir, args.patchOutput), + patchJson: path.relative(runDir, args.patchJsonOutput), + } + : {}), + }, + environment: { + runtime: isBunRuntime ? "bun" : "node", + platform: process.platform, + }, + result: { + status, + }, + }; + + fs.writeFileSync(path.join(runDir, "manifest.json"), `${JSON.stringify(manifest, null, 2)}\n`); +} + +function runPatchGate(args: Args): number { + if (!args.baselinePatch && !args.candidatePatch) return 0; + if (!args.baselinePatch || !args.candidatePatch) { + console.error("[router-eval] --baseline-patch and --candidate-patch must be provided together"); + return 2; + } + ensureReadable(args.baselinePatch, "baseline patch"); + ensureReadable(args.candidatePatch, "candidate patch"); + fs.mkdirSync(path.dirname(args.patchOutput), { recursive: true }); + fs.mkdirSync(path.dirname(args.patchJsonOutput), { recursive: true }); + + const result = runTypeScriptScript([ + "scripts/router-eval/patch-compare.ts", + "--baseline", + args.baselinePatch, + "--candidate", + args.candidatePatch, + "--output", + args.patchOutput, + "--json-output", + args.patchJsonOutput, + "--run-id", + args.runId, + "--max-aiq-drop", + args.maxPatchAiqDrop, + "--max-cost-increase", + args.maxPatchCostIncrease, + "--max-latency-increase", + args.maxPatchLatencyIncrease, + "--max-regression-increase", + args.maxPatchRegressionIncrease, + "--fail-on-regression", + ]); + + if (result.error) { + console.error(`[router-eval] failed to launch patch compare: ${result.error.message}`); + return 1; + } + if (result.stdout) process.stdout.write(result.stdout); + if (result.stderr) process.stderr.write(result.stderr); + return result.status ?? 1; +} + +function main(): void { + const args = readArgs(); + ensureReadable(args.baseline, "baseline corpus"); + ensureReadable(args.candidate, "candidate corpus"); + fs.mkdirSync(path.dirname(args.output), { recursive: true }); + fs.mkdirSync(path.dirname(args.jsonOutput), { recursive: true }); + + const result = runTypeScriptScript([ + "scripts/router-eval/index.ts", + "--input", + args.candidate, + "--baseline-input", + args.baseline, + "--max-aiq-drop", + args.maxAiqDrop, + "--max-cost-increase", + args.maxCostIncrease, + "--output", + args.output, + "--json-output", + args.jsonOutput, + "--fail-on-regression", + ]); + + if (result.error) { + console.error(`[router-eval] failed to launch evaluator: ${result.error.message}`); + process.exit(1); + } + + if (result.stdout) process.stdout.write(result.stdout); + if (result.stderr) process.stderr.write(result.stderr); + + if (result.status === 0) { + const patchStatus = runPatchGate(args); + writeRetainedRun(args, patchStatus); + if (patchStatus !== 0) { + console.error(`[router-eval] patch gate failed with exit code ${patchStatus}`); + process.exit(patchStatus); + } + const retention = args.artifactDir ? ` retained run ${args.runId}` : " temp run"; + console.log(`[router-eval] OK -${retention}; artifacts: ${args.output}, ${args.jsonOutput}`); + return; + } + + writeRetainedRun(args, result.status ?? 1); + console.error(`[router-eval] regression gate failed with exit code ${result.status ?? 1}`); + process.exit(result.status ?? 1); +} + +main(); diff --git a/scripts/router-eval/compare.ts b/scripts/router-eval/compare.ts new file mode 100644 index 00000000000..31ea7f36771 --- /dev/null +++ b/scripts/router-eval/compare.ts @@ -0,0 +1,135 @@ +#!/usr/bin/env node +import fs from "node:fs"; +import path from "node:path"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; + +type Args = { + baseline: string; + candidate: string; + baselineName: string; + candidateName: string; + artifactDir: string; + runId: string; + maxAiqDrop: string; + maxCostIncrease: string; + failOnRegression: boolean; +}; + +const repoRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); +const isBunRuntime = "Bun" in globalThis; + +function runTypeScriptScript(args: string[]) { + return spawnSync(process.execPath, isBunRuntime ? args : ["--import", "tsx", ...args], { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }); +} + +function getArgValue(name: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + if (index < 0) return undefined; + const value = process.argv[index + 1]; + return value && !value.startsWith("--") ? value : undefined; +} + +function usage(): string { + return [ + "Usage:", + " npm run eval:router:compare -- --baseline --candidate ", + " [--baseline-name ] [--candidate-name ] [--artifact-dir ]", + " [--run-id ] [--max-aiq-drop ] [--max-cost-increase ] [--fail-on-regression]", + "", + "Runs a named baseline-vs-candidate router-eval comparison and retains artifacts.", + ].join("\n"); +} + +function requireArg(name: string): string { + const value = getArgValue(name); + if (!value) { + console.error(`Missing required --${name}`); + process.exit(2); + } + return value; +} + +function readArgs(): Args { + const baselineName = getArgValue("baseline-name") ?? "baseline"; + const candidateName = getArgValue("candidate-name") ?? "candidate"; + const timestamp = new Date().toISOString().replace(/[:.]/g, "-"); + return { + baseline: requireArg("baseline"), + candidate: requireArg("candidate"), + baselineName, + candidateName, + artifactDir: getArgValue("artifact-dir") ?? "artifacts/router-eval/comparisons", + runId: getArgValue("run-id") ?? `${baselineName}-vs-${candidateName}-${timestamp}`, + maxAiqDrop: getArgValue("max-aiq-drop") ?? "1", + maxCostIncrease: getArgValue("max-cost-increase") ?? "0.05", + failOnRegression: process.argv.includes("--fail-on-regression"), + }; +} + +function ensureReadable(filePath: string, label: string): void { + if (!fs.existsSync(filePath)) { + console.error(`[router-eval:compare] ${label} missing: ${filePath}`); + process.exit(2); + } +} + +function main(): void { + if (process.argv.includes("--help") || process.argv.includes("-h")) { + console.log(usage()); + return; + } + + const args = readArgs(); + ensureReadable(args.baseline, "baseline corpus"); + ensureReadable(args.candidate, "candidate corpus"); + + const checkArgs = [ + "scripts/check/check-router-eval-regression.ts", + "--baseline", + args.baseline, + "--candidate", + args.candidate, + "--artifact-dir", + args.artifactDir, + "--run-id", + args.runId, + "--max-aiq-drop", + args.maxAiqDrop, + "--max-cost-increase", + args.maxCostIncrease, + ]; + + const result = runTypeScriptScript(checkArgs); + + if (result.error) { + console.error(`[router-eval:compare] failed to launch comparison: ${result.error.message}`); + process.exit(1); + } + + if (result.stdout) process.stdout.write(result.stdout); + if (result.stderr) process.stderr.write(result.stderr); + + const runDir = path.resolve(args.artifactDir, args.runId); + const labels = { + baselineName: args.baselineName, + candidateName: args.candidateName, + baseline: path.relative(runDir, path.resolve(args.baseline)), + candidate: path.relative(runDir, path.resolve(args.candidate)), + }; + fs.mkdirSync(runDir, { recursive: true }); + fs.writeFileSync(path.join(runDir, "comparison.json"), `${JSON.stringify(labels, null, 2)}\n`); + + if (result.status === 0 || !args.failOnRegression) { + console.log(`[router-eval:compare] artifacts: ${runDir}`); + return; + } + + process.exit(result.status ?? 1); +} + +main(); diff --git a/scripts/router-eval/index.ts b/scripts/router-eval/index.ts new file mode 100644 index 00000000000..6b0b7312e91 --- /dev/null +++ b/scripts/router-eval/index.ts @@ -0,0 +1,483 @@ +#!/usr/bin/env node +import fs from "node:fs"; +import path from "node:path"; + +import { + compareRouterEvalRuns, + createRouterEvalArtifact, + formatRouterEvalComparison, + formatRouterEvalReport, + runRouterEval, + toRouterObservation, + type RouterEvalArtifact, + type RouterEvalArtifactMetadata, + type RouterObservation, +} from "@/lib/routerEval/index.ts"; +import { SQLITE_FILE } from "@/lib/db/core.ts"; + +type DbCallLogRow = { + id: string; + model: string | null; + requested_model: string | null; + duration: number | null; + tokens_in: number | null; + tokens_out: number | null; + status: number | null; + combo_name: string | null; + provider: string | null; + error_summary: string | null; + timestamp: string | null; + correlation_id: string | null; +}; + +type DbUsageHistoryRow = { + id: number; + provider: string | null; + model: string | null; + tokens_input: number | null; + tokens_output: number | null; + service_tier: string | null; + status: string | null; + success: number | null; + latency_ms: number | null; + error_code: string | null; + combo_strategy: string | null; + timestamp: string | null; +}; + +type DbReplaySource = "auto" | "call-logs" | "usage-history"; + +type SqliteStatement = { + get: (...params: unknown[]) => unknown; + all: (...params: unknown[]) => unknown[]; +}; + +type SqliteDatabase = { + prepare: (sql: string) => SqliteStatement; + close: () => void; +}; + +type ArgSpec = { + input?: string; + db?: string; + dbSource?: DbReplaySource; + baselineInput?: string; + baselineDb?: string; + baselineDbSource?: DbReplaySource; + since?: string; + limit?: number; + aiqDrop?: number; + costIncrease?: number; + output?: string; + jsonOutput?: string; + exportCorpus?: string; + failOnRegression?: boolean; + help?: boolean; +}; + +function getArgValue(name: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + if (index < 0) return undefined; + const value = process.argv[index + 1]; + if (!value || value.startsWith("--")) return undefined; + return value; +} + +function getNumericArg(name: string): number | undefined { + const value = getArgValue(name); + if (!value) return undefined; + const parsed = Number.parseInt(value, 10); + return Number.isFinite(parsed) && parsed >= 0 ? parsed : undefined; +} + +function getFloatArg(name: string): number | undefined { + const value = getArgValue(name); + if (!value) return undefined; + const parsed = Number.parseFloat(value); + return Number.isFinite(parsed) ? parsed : undefined; +} + +function parseArgs(): ArgSpec { + return { + input: getArgValue("input"), + db: getArgValue("db"), + dbSource: parseReplaySource(getArgValue("db-source")), + baselineInput: getArgValue("baseline-input"), + baselineDb: getArgValue("baseline-db"), + baselineDbSource: parseReplaySource(getArgValue("baseline-db-source")), + since: getArgValue("since"), + limit: getNumericArg("limit"), + aiqDrop: getFloatArg("max-aiq-drop"), + costIncrease: getFloatArg("max-cost-increase"), + output: getArgValue("output"), + jsonOutput: getArgValue("json-output"), + exportCorpus: getArgValue("export-corpus"), + failOnRegression: process.argv.includes("--fail-on-regression"), + help: process.argv.includes("--help") || process.argv.includes("-h"), + }; +} + +function usage() { + return [ + "Usage:", + " npm run eval:router -- --input [--since ] [--limit ]", + " npm run eval:router -- --db [path] [--db-source usage-history|call-logs|auto] [--since ] [--limit ]", + " npm run eval:router -- --input --baseline-input ", + " npm run eval:router -- --db --db-source usage-history", + " [--max-aiq-drop ] [--max-cost-increase ] [--fail-on-regression]", + "", + "Options:", + " --input JSONL observation corpus (or omit for stdin)", + " --db [path] Read SQLite rows from the routing-replay source", + " --db-source Source for --db reads (default: auto => call-logs then usage-history)", + " --baseline-input Baseline corpus in JSONL", + " --baseline-db Baseline corpus in SQLite", + " --baseline-db-source Source for baseline DB reads", + " --since Filter rows newer than this value", + " --limit Limit sample count", + " --max-aiq-drop Regression threshold (default: 0)", + " --max-cost-increase Relative increase threshold (default: 0)", + " --output Write report to file", + " --json-output Write machine-readable artifact JSON", + " --export-corpus Write normalized RouterObservation JSONL", + " --fail-on-regression Exit 1 if candidate regresses vs baseline", + ].join("\n"); +} + +function parseInputLine(rawLine: string): RouterObservation | null { + const trimmed = rawLine.trim(); + if (!trimmed) return null; + try { + const parsed = JSON.parse(trimmed); + return toRouterObservation(parsed); + } catch { + return null; + } +} + +async function readJsonl(inputPath?: string): Promise { + let text: string; + if (!inputPath) { + text = await new Response(process.stdin, { duplex: "half" }).text(); + } else { + text = await fs.promises.readFile(path.resolve(inputPath), "utf8"); + } + + const observations: RouterObservation[] = []; + for (const line of text.split(/\r?\n/)) { + const parsed = parseInputLine(line); + if (parsed) observations.push(parsed); + } + return observations; +} + +function estimateCost(tokensIn: unknown, tokensOut: unknown): number { + const inTokens = typeof tokensIn === "number" ? tokensIn : 0; + const outTokens = typeof tokensOut === "number" ? tokensOut : 0; + return Number(((inTokens + outTokens) * 0.000001).toFixed(6)); +} + +function parseReplaySource(rawSource?: string): DbReplaySource { + if (!rawSource) return "auto"; + const normalized = rawSource.toLowerCase(); + if (normalized === "auto") return "auto"; + if (normalized === "usage_history" || normalized === "usage-history") return "usage-history"; + if (normalized === "call_logs" || normalized === "call-logs") return "call-logs"; + throw new Error(`Unsupported db source: ${rawSource}`); +} + +function hasReplayTable(database: SqliteDatabase, tableName: string): boolean { + return Boolean( + database.prepare("SELECT 1 FROM sqlite_master WHERE type='table' AND name=?").get(tableName) + ); +} + +function resolveReplaySource( + database: SqliteDatabase, + requestedSource: DbReplaySource +): DbReplaySource { + const hasUsageHistory = hasReplayTable(database, "usage_history"); + const hasCallLogs = hasReplayTable(database, "call_logs"); + + if (requestedSource === "usage-history") { + if (!hasUsageHistory) throw new Error("Table 'usage_history' missing in database"); + return "usage-history"; + } + + if (requestedSource === "call-logs") { + if (!hasCallLogs) throw new Error("Table 'call_logs' missing in database"); + return "call-logs"; + } + + if (requestedSource === "auto") { + if (hasCallLogs) return "call-logs"; + if (hasUsageHistory) return "usage-history"; + } + + throw new Error("No replay table found in database (expected usage_history or call_logs)"); +} + +function toSuccessFromStatus(status: unknown): boolean { + if (typeof status === "number") return status >= 200 && status < 400; + if (typeof status === "string") { + const parsed = Number.parseInt(status, 10); + if (Number.isFinite(parsed)) return parsed >= 200 && parsed < 400; + const normalized = status.trim().toLowerCase(); + if (normalized === "ok" || normalized === "success" || normalized === "true") return true; + } + return false; +} + +function readCallLogDb(db: SqliteDatabase, since?: string, limit?: number): RouterObservation[] { + const queryParts = [ + "SELECT id, model, requested_model, duration, tokens_in, tokens_out, status, combo_name, provider, error_summary, timestamp, correlation_id", + "FROM call_logs", + "WHERE 1=1", + ]; + const params: unknown[] = []; + + if (since) { + queryParts.push("AND timestamp >= ?"); + params.push(since); + } + + queryParts.push("ORDER BY timestamp ASC"); + if (limit) { + queryParts.push("LIMIT ?"); + params.push(limit); + } + + const rows = db.prepare(queryParts.join(" ")).all(...params) as DbCallLogRow[]; + + const observations: RouterObservation[] = []; + for (const row of rows) { + const mapped = toRouterObservation({ + sampleId: row.id, + model: row.model, + requestedModel: row.requested_model, + latency: row.duration ?? 0, + costUsd: estimateCost(row.tokens_in, row.tokens_out), + configId: row.combo_name || row.provider || "default", + success: row.status != null && row.status >= 200 && row.status < 400, + status: row.status ?? 0, + error: row.error_summary, + routeInput: { + correlationId: row.correlation_id ?? "", + }, + timestamp: row.timestamp ?? new Date().toISOString(), + }); + if (mapped) observations.push(mapped); + } + return observations; +} + +function readUsageHistoryDb( + db: SqliteDatabase, + since?: string, + limit?: number +): RouterObservation[] { + const queryParts = [ + "SELECT id, provider, model, tokens_input, tokens_output, service_tier, status, success, latency_ms, error_code, combo_strategy, timestamp", + "FROM usage_history", + "WHERE 1=1", + ]; + const params: unknown[] = []; + + if (since) { + queryParts.push("AND timestamp >= ?"); + params.push(since); + } + + queryParts.push("ORDER BY timestamp ASC"); + if (limit) { + queryParts.push("LIMIT ?"); + params.push(limit); + } + + const rows = db.prepare(queryParts.join(" ")).all(...params) as DbUsageHistoryRow[]; + + const observations: RouterObservation[] = []; + for (const row of rows) { + const cost = estimateCost(row.tokens_input, row.tokens_output); + const rowId = `${row.id}`; + const mapped = toRouterObservation({ + sampleId: rowId, + model: row.model, + requestedModel: row.model, + latency: row.latency_ms ?? 0, + costUsd: cost, + configId: row.combo_strategy || row.provider || "default", + success: toSuccessFromStatus(row.status) || row.success === 1, + status: row.success === 1 ? 200 : 0, + routeInput: {}, + metadata: { + provider: row.provider, + serviceTier: row.service_tier, + errorCode: row.error_code, + }, + timestamp: row.timestamp ?? new Date().toISOString(), + }); + if (mapped) observations.push(mapped); + } + return observations; +} + +async function openSqliteDatabase(sqliteFile: string): Promise { + if ("Bun" in globalThis) { + const sqlite = await import("bun:sqlite"); + return new sqlite.Database(sqliteFile, { readonly: true }); + } + + const sqlite = await import("better-sqlite3"); + return new sqlite.default(sqliteFile, { readonly: true }); +} + +async function readDb( + filePath: string, + since?: string, + limit?: number, + source: DbReplaySource = "auto" +): Promise { + const sqliteFile = filePath || SQLITE_FILE; + if (!sqliteFile) throw new Error("SQLite mode requires a path or SQLITE_FILE"); + const db = await openSqliteDatabase(sqliteFile); + try { + const normalized = parseReplaySource(source); + const activeSource = resolveReplaySource(db, normalized); + if (activeSource === "usage-history") { + return readUsageHistoryDb(db, since, limit); + } + return readCallLogDb(db, since, limit); + } finally { + db.close(); + } +} + +function resolveDbPath(rawArg?: string): string { + if (rawArg) return path.resolve(rawArg); + if (SQLITE_FILE) return SQLITE_FILE; + throw new Error("No SQLITE_FILE and no --db path provided"); +} + +function describeInputSource( + inputPath: string | undefined, + dbPath: string | undefined, + dbSource: DbReplaySource | undefined, + usesDb: boolean +): { source: string; path?: string; dbSource?: string } { + if (inputPath) return { source: "jsonl", path: path.resolve(inputPath) }; + if (usesDb) { + return { + source: "sqlite", + path: resolveDbPath(dbPath), + dbSource: dbSource ?? "auto", + }; + } + return { source: "stdin" }; +} + +function buildArtifactMetadata(args: ArgSpec, hasCandidateDb: boolean): RouterEvalArtifactMetadata { + const hasBaselineDb = Boolean(args.baselineDb); + return { + candidate: describeInputSource(args.input, args.db, args.dbSource, hasCandidateDb), + baseline: + args.baselineInput || hasBaselineDb + ? describeInputSource( + args.baselineInput, + args.baselineDb, + args.baselineDbSource, + hasBaselineDb + ) + : undefined, + window: { + since: args.since, + limit: args.limit, + }, + thresholds: { + maxAiqDrop: args.aiqDrop ?? 0, + maxCostIncrease: args.costIncrease ?? 0, + }, + outputs: { + markdown: args.output ? path.resolve(args.output) : undefined, + json: args.jsonOutput ? path.resolve(args.jsonOutput) : undefined, + corpus: args.exportCorpus ? path.resolve(args.exportCorpus) : undefined, + }, + }; +} + +async function writeCorpus(pathArg: string, observations: RouterObservation[]): Promise { + const outPath = path.resolve(pathArg); + await fs.promises.mkdir(path.dirname(outPath), { recursive: true }); + const lines = observations.map((observation) => JSON.stringify(observation)); + await fs.promises.writeFile(outPath, `${lines.join("\n")}\n`, "utf8"); +} + +async function run() { + const args = parseArgs(); + if (args.help) { + console.log(usage()); + return; + } + + const hasCandidateDb = Boolean(args.db || process.argv.includes("--db")); + const candidate: RouterObservation[] = args.input + ? await readJsonl(args.input) + : hasCandidateDb + ? await readDb(resolveDbPath(args.db), args.since, args.limit, args.dbSource) + : await readJsonl(); + + const baseline: RouterObservation[] | undefined = args.baselineInput + ? await readJsonl(args.baselineInput) + : args.baselineDb + ? await readDb(resolveDbPath(args.baselineDb), args.since, args.limit, args.baselineDbSource) + : undefined; + + if (candidate.length === 0) { + console.error("No candidate observations found"); + process.exitCode = 2; + return; + } + + if (args.exportCorpus) { + await writeCorpus(args.exportCorpus, candidate); + } + + const report = runRouterEval(candidate); + const metadata = buildArtifactMetadata(args, hasCandidateDb); + let output = formatRouterEvalReport(report); + let artifact: RouterEvalArtifact = createRouterEvalArtifact(report, metadata); + + if (baseline && baseline.length > 0) { + const comparison = compareRouterEvalRuns(runRouterEval(baseline), report, { + aiqDrop: args.aiqDrop ?? 0, + relativeCostIncrease: args.costIncrease ?? 0, + }); + output = formatRouterEvalComparison(comparison); + artifact = createRouterEvalArtifact(comparison, metadata); + console.log(output); + if (args.failOnRegression && comparison.regressions.length > 0) { + process.exitCode = 1; + } + } else { + console.log(output); + } + + if (args.output) { + const outPath = path.resolve(args.output); + await fs.promises.writeFile(outPath, output, "utf8"); + } + + if (args.jsonOutput) { + const outPath = path.resolve(args.jsonOutput); + await fs.promises.writeFile(outPath, `${JSON.stringify(artifact, null, 2)}\n`, "utf8"); + } +} + +run().catch((error) => { + if (error && typeof error === "object" && "message" in error) { + console.error((error as Error).message); + } else { + console.error(String(error)); + } + process.exitCode = 1; +}); diff --git a/scripts/router-eval/patch-compare.ts b/scripts/router-eval/patch-compare.ts new file mode 100644 index 00000000000..32fe80d9028 --- /dev/null +++ b/scripts/router-eval/patch-compare.ts @@ -0,0 +1,315 @@ +#!/usr/bin/env node +import fs from "node:fs"; +import path from "node:path"; + +type PatchOperation = { + op?: string; + path?: string; + value?: string; + evidence?: { + aiq?: number; + avgCostUsd?: number; + avgLatencyMs?: number; + regressions?: number; + }; + rationale?: string; +}; + +type RouterConfigPatchArtifact = { + schemaVersion?: number; + kind?: string; + generatedAt?: string; + applyPolicy?: string; + source?: { + objective?: string; + runId?: string; + artifactPath?: string; + }; + operations?: PatchOperation[]; +}; + +type PatchComparison = { + schemaVersion: 1; + kind: "router-config-patch-comparison"; + generatedAt: string; + runId: string; + thresholds: PatchThresholds; + baseline: PatchSummary; + candidate: PatchSummary; + delta: { + aiq: number; + avgCostUsd: number; + costIncreaseRatio: number; + avgLatencyMs: number; + latencyIncreaseRatio: number; + regressions: number; + }; + changedRecommendation: boolean; + regressions: string[]; + result: { + passed: boolean; + status: 0 | 1; + }; +}; + +type PatchThresholds = { + maxAiqDrop: number; + maxCostIncrease: number; + maxLatencyIncrease: number; + maxRegressionIncrease: number; +}; + +type PatchSummary = { + name: string; + file: string; + objective: string; + runId: string; + recommendedConfigId: string; + aiq: number; + avgCostUsd: number; + avgLatencyMs: number; + regressions: number; + applyPolicy: string; +}; + +function getArgValue(name: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + if (index < 0) return undefined; + const value = process.argv[index + 1]; + return value && !value.startsWith("--") ? value : undefined; +} + +function usage(): string { + return [ + "Usage:", + " npm run eval:router:patch-compare -- --baseline --candidate ", + " [--baseline-name ] [--candidate-name ] [--artifact-dir ] [--run-id ]", + " [--output ] [--json-output ] [--fail-on-regression]", + " [--max-aiq-drop ] [--max-cost-increase ] [--max-latency-increase ]", + " [--max-regression-increase ]", + "", + "Compares two retained router config patch proposals without applying them.", + ].join("\n"); +} + +function getNumberArg(name: string, fallback: number): number { + const value = getArgValue(name); + if (value === undefined) return fallback; + const parsed = Number(value); + if (!Number.isFinite(parsed) || parsed < 0) { + console.error(`Invalid --${name} value: ${value}`); + process.exit(2); + } + return parsed; +} + +function requireArg(name: string): string { + const value = getArgValue(name); + if (!value) { + console.error(`Missing required --${name}`); + process.exit(2); + } + return value; +} + +function readPatch(file: string): RouterConfigPatchArtifact { + if (!fs.existsSync(file)) { + console.error(`[router-eval:patch-compare] patch file missing: ${file}`); + process.exit(2); + } + try { + return JSON.parse(fs.readFileSync(file, "utf8")) as RouterConfigPatchArtifact; + } catch (error) { + console.error( + `[router-eval:patch-compare] invalid JSON in ${file}: ${(error as Error).message}` + ); + process.exit(2); + } +} + +function requireNumber(value: unknown, field: string, file: string): number { + if (typeof value !== "number" || !Number.isFinite(value)) { + console.error(`[router-eval:patch-compare] invalid numeric evidence field ${field} in ${file}`); + process.exit(2); + } + return value; +} + +function summarizePatch( + name: string, + file: string, + patch: RouterConfigPatchArtifact +): PatchSummary { + if (patch.kind !== "router-config-patch") { + console.error( + `[router-eval:patch-compare] invalid patch kind in ${file}: ${patch.kind ?? "missing"}` + ); + process.exit(2); + } + const operation = patch.operations?.[0]; + if (operation?.op !== "recommend-router-config") { + console.error( + `[router-eval:patch-compare] missing recommend-router-config operation in ${file}` + ); + process.exit(2); + } + if (typeof operation.value !== "string" || operation.value.length === 0) { + console.error(`[router-eval:patch-compare] invalid recommended config value in ${file}`); + process.exit(2); + } + return { + name, + file: path.resolve(file), + objective: patch.source?.objective ?? "unknown", + runId: patch.source?.runId ?? "unknown", + recommendedConfigId: operation.value, + aiq: requireNumber(operation.evidence?.aiq, "aiq", file), + avgCostUsd: requireNumber(operation.evidence?.avgCostUsd, "avgCostUsd", file), + avgLatencyMs: requireNumber(operation.evidence?.avgLatencyMs, "avgLatencyMs", file), + regressions: requireNumber(operation.evidence?.regressions, "regressions", file), + applyPolicy: patch.applyPolicy ?? "unknown", + }; +} + +function increaseRatio(delta: number, baseline: number): number { + if (baseline === 0) return delta > 0 ? Number.POSITIVE_INFINITY : 0; + return delta / baseline; +} + +function findRegressions( + baseline: PatchSummary, + candidate: PatchSummary, + thresholds: PatchThresholds +): string[] { + const aiqDrop = baseline.aiq - candidate.aiq; + const costDelta = candidate.avgCostUsd - baseline.avgCostUsd; + const latencyDelta = candidate.avgLatencyMs - baseline.avgLatencyMs; + const regressionDelta = candidate.regressions - baseline.regressions; + const regressions: string[] = []; + if (aiqDrop > thresholds.maxAiqDrop) regressions.push(`AIQ dropped by ${aiqDrop.toFixed(3)}`); + if (increaseRatio(costDelta, baseline.avgCostUsd) > thresholds.maxCostIncrease) { + regressions.push( + `average cost increased by ${increaseRatio(costDelta, baseline.avgCostUsd).toFixed(3)}` + ); + } + if (increaseRatio(latencyDelta, baseline.avgLatencyMs) > thresholds.maxLatencyIncrease) { + regressions.push( + `average latency increased by ${increaseRatio(latencyDelta, baseline.avgLatencyMs).toFixed(3)}` + ); + } + if (regressionDelta > thresholds.maxRegressionIncrease) { + regressions.push(`regression count increased by ${regressionDelta}`); + } + return regressions; +} + +function comparePatches( + runId: string, + thresholds: PatchThresholds, + failOnRegression: boolean, + baseline: PatchSummary, + candidate: PatchSummary +): PatchComparison { + const regressions = findRegressions(baseline, candidate, thresholds); + const status = failOnRegression && regressions.length > 0 ? 1 : 0; + const costDelta = candidate.avgCostUsd - baseline.avgCostUsd; + const latencyDelta = candidate.avgLatencyMs - baseline.avgLatencyMs; + return { + schemaVersion: 1, + kind: "router-config-patch-comparison", + generatedAt: new Date().toISOString(), + runId, + thresholds, + baseline, + candidate, + delta: { + aiq: candidate.aiq - baseline.aiq, + avgCostUsd: costDelta, + costIncreaseRatio: increaseRatio(costDelta, baseline.avgCostUsd), + avgLatencyMs: latencyDelta, + latencyIncreaseRatio: increaseRatio(latencyDelta, baseline.avgLatencyMs), + regressions: candidate.regressions - baseline.regressions, + }, + changedRecommendation: baseline.recommendedConfigId !== candidate.recommendedConfigId, + regressions, + result: { + passed: regressions.length === 0, + status, + }, + }; +} + +function formatComparison(comparison: PatchComparison): string { + return [ + "# Router Config Patch Comparison", + "", + `Passed: ${comparison.result.passed ? "yes" : "no"}`, + `Changed recommendation: ${comparison.changedRecommendation ? "yes" : "no"}`, + ...(comparison.regressions.length > 0 + ? ["", "## Regressions", "", ...comparison.regressions.map((item) => `- ${item}`)] + : []), + "", + "| Side | Name | Objective | Recommended Config | AIQ | Avg Cost | Avg Latency | Regressions | Apply Policy |", + "| --- | --- | --- | --- | ---: | ---: | ---: | ---: | --- |", + formatSummaryRow("Baseline", comparison.baseline), + formatSummaryRow("Candidate", comparison.candidate), + "", + "| Delta | AIQ | Avg Cost | Avg Latency | Regressions |", + "| --- | ---: | ---: | ---: | ---: |", + `| Candidate - Baseline | ${comparison.delta.aiq.toFixed(3)} | $${comparison.delta.avgCostUsd.toFixed(6)} | ${comparison.delta.avgLatencyMs.toFixed(2)}ms | ${comparison.delta.regressions} |`, + "", + ].join("\n"); +} + +function formatSummaryRow(side: string, summary: PatchSummary): string { + return `| ${side} | ${summary.name} | ${summary.objective} | ${summary.recommendedConfigId} | ${summary.aiq.toFixed(3)} | $${summary.avgCostUsd.toFixed(6)} | ${summary.avgLatencyMs.toFixed(2)}ms | ${summary.regressions} | ${summary.applyPolicy} |`; +} + +function main(): void { + if (process.argv.includes("--help") || process.argv.includes("-h")) { + console.log(usage()); + return; + } + + const baselineFile = requireArg("baseline"); + const candidateFile = requireArg("candidate"); + const baselineName = getArgValue("baseline-name") ?? "baseline"; + const candidateName = getArgValue("candidate-name") ?? "candidate"; + const runId = getArgValue("run-id") ?? new Date().toISOString().replace(/[:.]/g, "-"); + const thresholds: PatchThresholds = { + maxAiqDrop: getNumberArg("max-aiq-drop", Number.POSITIVE_INFINITY), + maxCostIncrease: getNumberArg("max-cost-increase", Number.POSITIVE_INFINITY), + maxLatencyIncrease: getNumberArg("max-latency-increase", Number.POSITIVE_INFINITY), + maxRegressionIncrease: getNumberArg("max-regression-increase", Number.POSITIVE_INFINITY), + }; + const failOnRegression = process.argv.includes("--fail-on-regression"); + const baseline = summarizePatch(baselineName, baselineFile, readPatch(baselineFile)); + const candidate = summarizePatch(candidateName, candidateFile, readPatch(candidateFile)); + const comparison = comparePatches(runId, thresholds, failOnRegression, baseline, candidate); + const markdown = formatComparison(comparison); + const artifactDir = getArgValue("artifact-dir"); + const output = getArgValue("output"); + const jsonOutput = getArgValue("json-output"); + + if (artifactDir) { + const runDir = path.resolve(artifactDir, runId); + fs.mkdirSync(runDir, { recursive: true }); + fs.writeFileSync(path.join(runDir, "patch-comparison.md"), markdown); + fs.writeFileSync( + path.join(runDir, "patch-comparison.json"), + `${JSON.stringify(comparison, null, 2)}\n` + ); + } + if (output) { + fs.mkdirSync(path.dirname(path.resolve(output)), { recursive: true }); + fs.writeFileSync(output, markdown); + } + if (jsonOutput) { + fs.mkdirSync(path.dirname(path.resolve(jsonOutput)), { recursive: true }); + fs.writeFileSync(jsonOutput, `${JSON.stringify(comparison, null, 2)}\n`); + } + console.log(markdown); + process.exit(comparison.result.status); +} + +main(); diff --git a/scripts/router-eval/search.ts b/scripts/router-eval/search.ts new file mode 100644 index 00000000000..1683585e558 --- /dev/null +++ b/scripts/router-eval/search.ts @@ -0,0 +1,439 @@ +#!/usr/bin/env node +import fs from "node:fs"; +import path from "node:path"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; + +import type { RouterConfigAggregate, RouterEvalArtifact } from "@/lib/routerEval/index.ts"; + +type Candidate = { + name: string; + path: string; +}; + +type SearchResult = { + candidateName: string; + configId: string; + runId: string; + aiq: number; + avgCostUsd: number; + avgLatencyMs: number; + regressions: number; + artifactPath: string; +}; + +type SearchObjective = "balanced" | "quality" | "cost" | "latency"; + +type SearchRecommendation = SearchResult & { + objective: SearchObjective; + rank: number; + rationale: string; +}; + +type RouterConfigSuggestion = { + schemaVersion: 1; + kind: "router-config-suggestion"; + generatedAt: string; + objective: SearchObjective; + recommendedConfigId: string; + sourceRunId: string; + sourceArtifactPath: string; + evidence: { + aiq: number; + avgCostUsd: number; + avgLatencyMs: number; + regressions: number; + }; + applyPolicy: "manual-review"; + rationale: string; +}; + +type RouterConfigPatchArtifact = { + schemaVersion: 1; + kind: "router-config-patch"; + generatedAt: string; + applyPolicy: "manual-review"; + source: { + objective: SearchObjective; + runId: string; + artifactPath: string; + }; + operations: Array<{ + op: "recommend-router-config"; + path: "/router/recommendedConfigId"; + value: string; + evidence: RouterConfigSuggestion["evidence"]; + rationale: string; + }>; +}; + +const repoRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); +const isBunRuntime = "Bun" in globalThis; + +function runTypeScriptScript(args: string[]) { + return spawnSync(process.execPath, isBunRuntime ? args : ["--import", "tsx", ...args], { + cwd: repoRoot, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }); +} + +function getArgValue(name: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + if (index < 0) return undefined; + const value = process.argv[index + 1]; + return value && !value.startsWith("--") ? value : undefined; +} + +function getArgValues(name: string): string[] { + const values: string[] = []; + for (let index = 0; index < process.argv.length; index += 1) { + if (process.argv[index] !== `--${name}`) continue; + const value = process.argv[index + 1]; + if (value && !value.startsWith("--")) values.push(value); + } + return values; +} + +function usage(): string { + return [ + "Usage:", + " npm run eval:router:search -- --baseline ", + " --candidate [--candidate ...]", + " [--objective balanced|quality|cost|latency]", + " [--artifact-dir ] [--run-id ] [--max-aiq-drop ] [--max-cost-increase ]", + "", + "Ranks candidate corpora by router-eval AIQ while retaining comparison artifacts.", + ].join("\n"); +} + +function requireArg(name: string): string { + const value = getArgValue(name); + if (!value) { + console.error(`Missing required --${name}`); + process.exit(2); + } + return value; +} + +function parseCandidate(raw: string): Candidate { + const splitAt = raw.indexOf("="); + if (splitAt <= 0 || splitAt === raw.length - 1) { + console.error(`Invalid --candidate value: ${raw}. Expected name=path.ndjson`); + process.exit(2); + } + return { + name: raw.slice(0, splitAt), + path: raw.slice(splitAt + 1), + }; +} + +function ensureReadable(filePath: string, label: string): void { + if (!fs.existsSync(filePath)) { + console.error(`[router-eval:search] ${label} missing: ${filePath}`); + process.exit(2); + } +} + +function readArtifact(filePath: string): RouterEvalArtifact { + return JSON.parse(fs.readFileSync(filePath, "utf8")) as RouterEvalArtifact; +} + +function parseObjective(value: string | undefined): SearchObjective { + if (!value) return "balanced"; + if (value === "balanced" || value === "quality" || value === "cost" || value === "latency") { + return value; + } + console.error( + `Invalid --objective value: ${value}. Expected balanced, quality, cost, or latency.` + ); + process.exit(2); +} + +function compareConfigsByObjective( + objective: SearchObjective, + a: RouterConfigAggregate, + b: RouterConfigAggregate +): number { + if (objective === "cost") { + return a.avgCostUsd - b.avgCostUsd || b.aiq - a.aiq || a.avgLatencyMs - b.avgLatencyMs; + } + if (objective === "latency") { + return a.avgLatencyMs - b.avgLatencyMs || b.aiq - a.aiq || a.avgCostUsd - b.avgCostUsd; + } + return b.aiq - a.aiq || a.avgCostUsd - b.avgCostUsd || a.avgLatencyMs - b.avgLatencyMs; +} + +function selectBestConfig( + artifact: RouterEvalArtifact, + objective: SearchObjective +): RouterConfigAggregate | undefined { + const configs = + artifact.comparison?.candidate.configurations ?? + artifact.report?.configurations ?? + artifact.report?.top ?? + []; + return [...configs].sort((a, b) => compareConfigsByObjective(objective, a, b))[0]; +} + +function resultFromArtifact( + candidateName: string, + runId: string, + artifactPath: string, + objective: SearchObjective +): SearchResult { + const artifact = readArtifact(artifactPath); + const best = selectBestConfig(artifact, objective); + if (!best) { + throw new Error(`No best candidate found in ${artifactPath}`); + } + return { + candidateName, + configId: best.configId, + runId, + aiq: best.aiq, + avgCostUsd: best.avgCostUsd, + avgLatencyMs: best.avgLatencyMs, + regressions: artifact.comparison?.regressions.length ?? 0, + artifactPath, + }; +} + +function formatSearch(results: SearchResult[]): string { + const lines = [ + "# Router Eval Search", + "", + "| Rank | Candidate | AIQ | Avg Cost | Avg Latency | Regressions | Run |", + "| ---: | --- | ---: | ---: | ---: | ---: | --- |", + ]; + results.forEach((result, index) => { + lines.push( + `| ${index + 1} | ${result.candidateName} | ${result.aiq.toFixed(3)} | $${result.avgCostUsd.toFixed(6)} | ${result.avgLatencyMs.toFixed(2)}ms | ${result.regressions} | ${result.runId} |` + ); + }); + return `${lines.join("\n")}\n`; +} + +function compareByObjective(objective: SearchObjective, a: SearchResult, b: SearchResult): number { + if (objective === "cost") { + return ( + a.regressions - b.regressions || + a.avgCostUsd - b.avgCostUsd || + b.aiq - a.aiq || + a.avgLatencyMs - b.avgLatencyMs + ); + } + if (objective === "latency") { + return ( + a.regressions - b.regressions || + a.avgLatencyMs - b.avgLatencyMs || + b.aiq - a.aiq || + a.avgCostUsd - b.avgCostUsd + ); + } + if (objective === "quality") { + return ( + b.aiq - a.aiq || + a.regressions - b.regressions || + a.avgCostUsd - b.avgCostUsd || + a.avgLatencyMs - b.avgLatencyMs + ); + } + return ( + b.aiq - a.aiq || + a.regressions - b.regressions || + a.avgCostUsd - b.avgCostUsd || + a.avgLatencyMs - b.avgLatencyMs + ); +} + +function recommendationRationale(objective: SearchObjective, result: SearchResult): string { + if (objective === "cost") { + return `${result.candidateName} has the best cost-first rank with ${result.regressions} regressions and $${result.avgCostUsd.toFixed(6)} average cost.`; + } + if (objective === "latency") { + return `${result.candidateName} has the best latency-first rank with ${result.regressions} regressions and ${result.avgLatencyMs.toFixed(2)}ms average latency.`; + } + if (objective === "quality") { + return `${result.candidateName} has the best quality-first rank with ${result.aiq.toFixed(3)} AIQ.`; + } + return `${result.candidateName} has the best balanced rank with ${result.aiq.toFixed(3)} AIQ, ${result.regressions} regressions, $${result.avgCostUsd.toFixed(6)} average cost, and ${result.avgLatencyMs.toFixed(2)}ms average latency.`; +} + +function createRecommendation( + objective: SearchObjective, + results: SearchResult[] +): SearchRecommendation { + const winner = results[0]; + if (!winner) { + throw new Error("Cannot create a recommendation without search results"); + } + return { + ...winner, + objective, + rank: 1, + rationale: recommendationRationale(objective, winner), + }; +} + +function createConfigSuggestion( + generatedAt: string, + recommendation: SearchRecommendation +): RouterConfigSuggestion { + return { + schemaVersion: 1, + kind: "router-config-suggestion", + generatedAt, + objective: recommendation.objective, + recommendedConfigId: recommendation.configId, + sourceRunId: recommendation.runId, + sourceArtifactPath: recommendation.artifactPath, + evidence: { + aiq: recommendation.aiq, + avgCostUsd: recommendation.avgCostUsd, + avgLatencyMs: recommendation.avgLatencyMs, + regressions: recommendation.regressions, + }, + applyPolicy: "manual-review", + rationale: recommendation.rationale, + }; +} + +function createConfigPatch(suggestion: RouterConfigSuggestion): RouterConfigPatchArtifact { + return { + schemaVersion: 1, + kind: "router-config-patch", + generatedAt: suggestion.generatedAt, + applyPolicy: "manual-review", + source: { + objective: suggestion.objective, + runId: suggestion.sourceRunId, + artifactPath: suggestion.sourceArtifactPath, + }, + operations: [ + { + op: "recommend-router-config", + path: "/router/recommendedConfigId", + value: suggestion.recommendedConfigId, + evidence: suggestion.evidence, + rationale: suggestion.rationale, + }, + ], + }; +} + +function formatPatchOperations(patch: RouterConfigPatchArtifact): string { + const lines = [ + "## Patch Operations", + "", + "| Op | Path | Value | AIQ | Avg Cost | Avg Latency | Regressions | Apply Policy |", + "| --- | --- | --- | ---: | ---: | ---: | ---: | --- |", + ]; + for (const operation of patch.operations) { + lines.push( + `| ${operation.op} | ${operation.path} | ${operation.value} | ${operation.evidence.aiq.toFixed(3)} | $${operation.evidence.avgCostUsd.toFixed(6)} | ${operation.evidence.avgLatencyMs.toFixed(2)}ms | ${operation.evidence.regressions} | ${patch.applyPolicy} |` + ); + } + return `${lines.join("\n")}\n`; +} + +function main(): void { + if (process.argv.includes("--help") || process.argv.includes("-h")) { + console.log(usage()); + return; + } + + const baseline = requireArg("baseline"); + ensureReadable(baseline, "baseline corpus"); + const candidates = getArgValues("candidate").map(parseCandidate); + if (candidates.length === 0) { + console.error("At least one --candidate is required"); + process.exit(2); + } + + const artifactDir = getArgValue("artifact-dir") ?? "artifacts/router-eval/search"; + const searchId = getArgValue("run-id") ?? new Date().toISOString().replace(/[:.]/g, "-"); + const objective = parseObjective(getArgValue("objective")); + const maxAiqDrop = getArgValue("max-aiq-drop") ?? "1"; + const maxCostIncrease = getArgValue("max-cost-increase") ?? "0.05"; + const searchDir = path.resolve(artifactDir, searchId); + fs.mkdirSync(searchDir, { recursive: true }); + + const results: SearchResult[] = []; + for (const candidate of candidates) { + ensureReadable(candidate.path, `${candidate.name} corpus`); + const runId = `${searchId}-${candidate.name}`; + const result = runTypeScriptScript([ + "scripts/router-eval/compare.ts", + "--baseline", + baseline, + "--candidate", + candidate.path, + "--baseline-name", + "baseline", + "--candidate-name", + candidate.name, + "--artifact-dir", + searchDir, + "--run-id", + runId, + "--max-aiq-drop", + maxAiqDrop, + "--max-cost-increase", + maxCostIncrease, + ]); + if (result.stdout) process.stdout.write(result.stdout); + if (result.stderr) process.stderr.write(result.stderr); + if (result.error) { + console.error( + `[router-eval:search] failed to launch ${candidate.name}: ${result.error.message}` + ); + process.exit(1); + } + if (result.status !== 0) { + console.error( + `[router-eval:search] comparison failed for ${candidate.name} with exit code ${result.status ?? 1}` + ); + process.exit(result.status ?? 1); + } + const artifactPath = path.join(searchDir, runId, "router-eval.json"); + results.push(resultFromArtifact(candidate.name, runId, artifactPath, objective)); + } + + results.sort((a, b) => compareByObjective(objective, a, b)); + + const recommendation = createRecommendation(objective, results); + const generatedAt = new Date().toISOString(); + const suggestion = createConfigSuggestion(generatedAt, recommendation); + const patch = createConfigPatch(suggestion); + const markdown = `${formatSearch(results)}## Recommendation\n\n${recommendation.rationale}\n\n${formatPatchOperations(patch)}`; + const summary = { + schemaVersion: 1, + kind: "router-eval-search", + generatedAt, + baseline: path.resolve(baseline), + objective, + recommendation, + suggestion, + patch, + results, + }; + fs.writeFileSync(path.join(searchDir, "search.md"), markdown); + fs.writeFileSync(path.join(searchDir, "search.json"), `${JSON.stringify(summary, null, 2)}\n`); + fs.writeFileSync( + path.join(searchDir, "recommendation.json"), + `${JSON.stringify(recommendation, null, 2)}\n` + ); + fs.writeFileSync( + path.join(searchDir, "suggestion.json"), + `${JSON.stringify(suggestion, null, 2)}\n` + ); + fs.writeFileSync( + path.join(searchDir, "router-config.patch.json"), + `${JSON.stringify(patch, null, 2)}\n` + ); + console.log(markdown); + console.log(`[router-eval:search] artifacts: ${searchDir}`); +} + +main(); diff --git a/scripts/router-eval/trends.ts b/scripts/router-eval/trends.ts new file mode 100644 index 00000000000..70f277c1100 --- /dev/null +++ b/scripts/router-eval/trends.ts @@ -0,0 +1,198 @@ +#!/usr/bin/env node +import fs from "node:fs"; +import path from "node:path"; + +import type { + RouterConfigAggregate, + RouterEvalArtifact, + RouterEvalArtifactMetadata, + RouterEvalComparison, + RouterEvalReport, +} from "@/lib/routerEval/index.ts"; + +type TrendRow = { + runId: string; + generatedAt: string; + kind: RouterEvalArtifact["kind"]; + bestConfig: string; + aiq: number; + avgCostUsd: number; + avgLatencyMs: number; + regressions: number; + source: string; + window: string; +}; + +function getArgValue(name: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + if (index < 0) return undefined; + const value = process.argv[index + 1]; + return value && !value.startsWith("--") ? value : undefined; +} + +function usage(): string { + return [ + "Usage:", + " npm run eval:router:trends -- --artifact-dir [--limit ] [--dashboard]", + "", + "Reads retained router-eval JSON artifacts and prints a markdown trend table or dashboard.", + ].join("\n"); +} + +function readJson(filePath: string): RouterEvalArtifact | null { + try { + return JSON.parse(fs.readFileSync(filePath, "utf8")) as RouterEvalArtifact; + } catch { + return null; + } +} + +function bestFromReport(report: RouterEvalReport): RouterConfigAggregate | undefined { + return report.top[0]; +} + +function bestFromComparison(comparison: RouterEvalComparison): RouterConfigAggregate | undefined { + return comparison.candidate.top[0]; +} + +function toTrendRow(runId: string, artifact: RouterEvalArtifact): TrendRow | null { + const best = artifact.comparison + ? bestFromComparison(artifact.comparison) + : artifact.report + ? bestFromReport(artifact.report) + : undefined; + + if (!best) return null; + + return { + runId, + generatedAt: artifact.generatedAt, + kind: artifact.kind, + bestConfig: best.configId, + aiq: best.aiq, + avgCostUsd: best.avgCostUsd, + avgLatencyMs: best.avgLatencyMs, + regressions: artifact.comparison?.regressions.length ?? 0, + source: artifact.metadata?.candidate?.source ?? "unknown", + window: formatWindow(artifact.metadata?.window), + }; +} + +function formatWindow(window: RouterEvalArtifactMetadata["window"]): string { + if (!window || typeof window !== "object") return "all"; + const parts: string[] = []; + if ("since" in window && typeof window.since === "string") parts.push(`since ${window.since}`); + if ("limit" in window && typeof window.limit === "number") parts.push(`limit ${window.limit}`); + return parts.length > 0 ? parts.join(", ") : "all"; +} + +function collectTrendRows(artifactDir: string): TrendRow[] { + if (!fs.existsSync(artifactDir)) return []; + + const rows: TrendRow[] = []; + for (const entry of fs.readdirSync(artifactDir, { withFileTypes: true })) { + const runId = entry.name; + const jsonPath = entry.isDirectory() + ? path.join(artifactDir, runId, "router-eval.json") + : entry.isFile() && entry.name.endsWith(".json") + ? path.join(artifactDir, entry.name) + : ""; + if (!jsonPath) continue; + + const artifact = readJson(jsonPath); + if (!artifact || artifact.schemaVersion !== 1) continue; + const row = toTrendRow(runId.replace(/\.json$/, ""), artifact); + if (row) rows.push(row); + } + + return rows.sort((a, b) => a.generatedAt.localeCompare(b.generatedAt)); +} + +function formatTrend(rows: TrendRow[], limit: number): string { + const limited = rows.slice(-limit); + const lines = [ + "# Router Eval Trends", + "", + "| Run | Kind | Source | Window | Best Config | AIQ | Avg Cost | Avg Latency | Regressions |", + "| --- | --- | --- | --- | --- | ---: | ---: | ---: | ---: |", + ]; + + for (const row of limited) { + lines.push( + `| ${row.runId} | ${row.kind} | ${row.source} | ${row.window} | ${row.bestConfig} | ${row.aiq.toFixed(3)} | $${row.avgCostUsd.toFixed(6)} | ${row.avgLatencyMs.toFixed(2)}ms | ${row.regressions} |` + ); + } + + return `${lines.join("\n")}\n`; +} + +function formatDelta(value: number): string { + if (value > 0) return `+${value.toFixed(3)}`; + return value.toFixed(3); +} + +function average(values: number[]): number { + if (values.length === 0) return 0; + return values.reduce((sum, value) => sum + value, 0) / values.length; +} + +function formatDashboard(rows: TrendRow[], limit: number): string { + const limited = rows.slice(-limit); + const latest = limited[limited.length - 1]; + const previous = limited[limited.length - 2]; + const aiqDelta = latest && previous ? latest.aiq - previous.aiq : 0; + const latencyDelta = latest && previous ? latest.avgLatencyMs - previous.avgLatencyMs : 0; + const costDelta = latest && previous ? latest.avgCostUsd - previous.avgCostUsd : 0; + const regressions = limited.reduce((sum, row) => sum + row.regressions, 0); + const lines = [ + "# Router Eval Dashboard", + "", + `Runs: ${limited.length}`, + `Latest: ${latest?.runId ?? "n/a"}`, + `Best config: ${latest?.bestConfig ?? "n/a"}`, + `AIQ: ${latest ? latest.aiq.toFixed(3) : "0.000"} (${formatDelta(aiqDelta)})`, + `Avg latency: ${latest ? latest.avgLatencyMs.toFixed(2) : "0.00"}ms (${formatDelta(latencyDelta)}ms)`, + `Avg cost: $${latest ? latest.avgCostUsd.toFixed(6) : "0.000000"} (${formatDelta(costDelta)})`, + `Window: ${latest?.window ?? "all"}`, + `Source: ${latest?.source ?? "unknown"}`, + `Regression count: ${regressions}`, + "", + "## Rolling Averages", + "", + `AIQ: ${average(limited.map((row) => row.aiq)).toFixed(3)}`, + `Latency: ${average(limited.map((row) => row.avgLatencyMs)).toFixed(2)}ms`, + `Cost: $${average(limited.map((row) => row.avgCostUsd)).toFixed(6)}`, + ]; + + return `${lines.join("\n")}\n`; +} + +function main(): void { + if (process.argv.includes("--help") || process.argv.includes("-h")) { + console.log(usage()); + return; + } + + const artifactDir = getArgValue("artifact-dir"); + if (!artifactDir) { + console.error("Missing required --artifact-dir"); + process.exit(2); + } + + const limit = Number.parseInt(getArgValue("limit") ?? "20", 10); + const rows = collectTrendRows(path.resolve(artifactDir)); + if (rows.length === 0) { + console.error(`No router-eval artifacts found in ${artifactDir}`); + process.exit(2); + } + + const boundedLimit = Number.isFinite(limit) && limit > 0 ? limit : 20; + if (process.argv.includes("--dashboard")) { + console.log(formatDashboard(rows, boundedLimit)); + return; + } + + console.log(formatTrend(rows, boundedLimit)); +} + +main(); diff --git a/skills/omni-context-rtk/SKILL.md b/skills/omni-context-rtk/SKILL.md index 8cb4d8193fd..8bad8466e57 100644 --- a/skills/omni-context-rtk/SKILL.md +++ b/skills/omni-context-rtk/SKILL.md @@ -43,6 +43,17 @@ curl https://localhost:20128/api/context/rtk/filters \ -H "Authorization: Bearer $OMNIROUTE_TOKEN" ``` +### POST /api/context/rtk/import + +Validate or install an RTK TOML schema v1 filter file + +```bash +curl -X POST https://localhost:20128/api/context/rtk/import \ + -H "Authorization: Bearer $OMNIROUTE_TOKEN" + -H "Content-Type: application/json" \ + -d '{}' +``` + ### POST /api/context/rtk/test Run RTK compression preview for text diff --git a/src/app/(dashboard)/dashboard/HomeProviderTopologySection.tsx b/src/app/(dashboard)/dashboard/HomeProviderTopologySection.tsx index 53abbbeed21..8aef005e07a 100644 --- a/src/app/(dashboard)/dashboard/HomeProviderTopologySection.tsx +++ b/src/app/(dashboard)/dashboard/HomeProviderTopologySection.tsx @@ -27,9 +27,14 @@ export function HomeProviderTopologySection({ enabled?: boolean; }) { const t = useTranslations("home"); + const tCommon = useTranslations("common"); + const tSettings = useTranslations("settings"); + const tAnalytics = useTranslations("analytics"); // #4596: gate the live-WS connection so it only opens while the topology // section is actually shown on the home page. const { activeRequests: liveActiveRequests } = useLiveRequests({ enabled }); + const activeRequests = selectActiveRequests(liveActiveRequests); + const activeProviderCount = new Set(activeRequests.map(({ provider }) => provider)).size; return ( @@ -37,24 +42,27 @@ export function HomeProviderTopologySection({

{t("providerTopology")}

- Connected providers routing through OmniRoute in real time + {t("activeError", { active: activeProviderCount, errors: errorProvider ? 1 : 0 })}

- Active + + {tCommon("active")} - Recent + + {tSettings("recent")} - Error + + {tAnalytics("modelStatusError")}
diff --git a/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx b/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx index 1ad0ea99267..c5e9c4d7f11 100644 --- a/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx +++ b/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx @@ -83,7 +83,10 @@ function ModeBar({ {skipped > 0 && ( // #4268: attempted-but-no-op runs (e.g. Stacked saved nothing) are // recorded now, so this mode is visible even when count is 0. - · {skipped.toLocaleString()} skipped (no-op) + + {" "} + · {skipped.toLocaleString()} skipped (no-op) + )} @@ -174,6 +177,7 @@ export default function CompressionAnalyticsTab() { const modes = Object.entries(stats.byMode).sort(([, a], [, b]) => b.count - a.count); const providers = Object.entries(stats.byProvider).sort(([, a], [, b]) => b.count - a.count); + const totalAttempts = stats.totalRequests + (stats.totalSkipped ?? 0); // Calculate max tokens for hourly chart scaling const maxTokensPerHour = Math.max(...stats.last24h.map((h) => h.tokensSaved), 1); @@ -209,7 +213,7 @@ export default function CompressionAnalyticsTab() { compress diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 4efc9618a3c..ea29735d058 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -1,6 +1,6 @@ "use client"; -import { useState, useEffect, useCallback, useMemo, useRef } from "react"; +import { useState, useEffect, useCallback, useMemo, useRef, memo } from "react"; import dynamic from "next/dynamic"; import Link from "next/link"; import { useRouter, useSearchParams } from "next/navigation"; @@ -1549,7 +1549,7 @@ function ComboReadinessPanel({ checks, blockers, showDescription = true }) { ); } -function ComboCard({ +function ComboCardInner({ combo, metrics, compressionEnabled, @@ -1767,6 +1767,7 @@ function ComboCard({
); } +const ComboCard = memo(ComboCardInner); function TestResultsView({ results }) { const emailsVisible = useEmailPrivacyStore((s) => s.emailsVisible); diff --git a/src/app/(dashboard)/dashboard/context/rtk/RtkContextPageClient.tsx b/src/app/(dashboard)/dashboard/context/rtk/RtkContextPageClient.tsx index 13df110d249..c49f89b70f6 100644 --- a/src/app/(dashboard)/dashboard/context/rtk/RtkContextPageClient.tsx +++ b/src/app/(dashboard)/dashboard/context/rtk/RtkContextPageClient.tsx @@ -4,6 +4,7 @@ import { useEffect, useMemo, useState } from "react"; import { useTranslations } from "next-intl"; import { SegmentedControl, Collapsible } from "@/shared/components"; import RtkLearnDiscoverCard from "./RtkLearnDiscoverCard"; +import RtkTomlImportCard from "./RtkTomlImportCard"; type RtkFilter = { id: string; @@ -78,11 +79,14 @@ export default function RtkContextPageClient() { .catch(() => {}); }, []); - useEffect(() => { + const loadFilters = () => fetch("/api/context/rtk/filters") .then((res) => (res.ok ? res.json() : null)) .then((data) => setFilters(Array.isArray(data?.filters) ? data.filters : [])) .catch(() => {}); + + useEffect(() => { + void loadFilters(); fetch("/api/context/rtk/config") .then((res) => (res.ok ? res.json() : null)) .then((data) => setConfig(data)) @@ -363,6 +367,8 @@ export default function RtkContextPageClient() { + {viewMode === "advanced" && } + ); diff --git a/src/app/(dashboard)/dashboard/context/rtk/RtkTomlImportCard.tsx b/src/app/(dashboard)/dashboard/context/rtk/RtkTomlImportCard.tsx new file mode 100644 index 00000000000..f6a3718bbf7 --- /dev/null +++ b/src/app/(dashboard)/dashboard/context/rtk/RtkTomlImportCard.tsx @@ -0,0 +1,255 @@ +"use client"; + +import { useState } from "react"; +import { useTranslations } from "next-intl"; + +const RTK_TOML_MAX_BYTES = 1024 * 1024; + +interface ImportFilterSummary { + id: string; + description: string; + category: string; + commandPatterns: string[]; + testCount: number; +} + +interface ImportTestOutcome { + filterId: string; + testName: string; + passed: boolean; +} + +interface ImportResult { + sha256: string; + passed: boolean; + filters: ImportFilterSummary[]; + outcomes: ImportTestOutcome[]; + warnings: string[]; + installedPath?: string; + backupCreated?: boolean; +} + +interface RtkTomlImportCardProps { + onInstalled?: () => void | Promise; +} + +interface RtkTomlEditorProps { + content: string; + processing: "validate" | "install" | null; + overwrite: boolean; + onContentChange: (content: string) => void; + onFileChange: (file: File | undefined) => void; + onProcess: (action: "validate" | "install") => void; + onOverwriteChange: (overwrite: boolean) => void; +} + +async function readErrorMessage(response: Response): Promise { + try { + const body = (await response.json()) as { error?: { message?: unknown } }; + return typeof body.error?.message === "string" ? body.error.message : null; + } catch { + return null; + } +} + +function RtkTomlEditor({ + content, + processing, + overwrite, + onContentChange, + onFileChange, + onProcess, + onOverwriteChange, +}: RtkTomlEditorProps) { + const t = useTranslations("contextRtk"); + return ( + <> +
+ +