From 11952ae0fee5baeae5346e1dee1e9639445f7c8b Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:40:08 -0300 Subject: [PATCH 01/18] feat(compression): add ultraEngine + ultraSlmPrewarm to CompressionConfig (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- open-sse/services/compression/types.ts | 14 ++++++++++++++ tests/unit/compression/ultra-slm-tier.test.ts | 15 ++++++++++++++- 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/open-sse/services/compression/types.ts b/open-sse/services/compression/types.ts index 14e25ef24d9..24b4ac8416e 100644 --- a/open-sse/services/compression/types.ts +++ b/open-sse/services/compression/types.ts @@ -161,6 +161,18 @@ export interface CompressionConfig { * change for installs that predate the panel). Set by `getCompressionSettings`. */ enginesExplicit?: boolean; + /** + * Phase 4 (B): which tier the `ultra` mode uses. + * "heuristic" = Tier-A token pruner (`pruneByScore`, default, byte-identical to pre-B). + * "slm" = Tier-B LLMLingua-2 ONNX worker when available, else fail-open to Tier-A. + */ + ultraEngine?: "heuristic" | "slm"; + /** + * Phase 4 (B): best-effort pre-warm of the SLM model on the enable transition + * and on a cold restart when `ultraEngine: "slm"` is already set. Failures are + * swallowed; the lazy first-call path still applies. Default false. + */ + ultraSlmPrewarm?: boolean; } export interface CompressionStats { @@ -229,6 +241,8 @@ export const DEFAULT_COMPRESSION_CONFIG: CompressionConfig = { ], engines: Object.fromEntries(ENGINE_IDS.map((id) => [id, { enabled: false }])), activeComboId: null, + ultraEngine: "heuristic", + ultraSlmPrewarm: false, }; export const DEFAULT_CAVEMAN_CONFIG: CavemanConfig = { diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index 8db4c37769a..bea1503362a 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -10,7 +10,7 @@ * The llmlingua backend is injectable (setLlmlinguaBackend), so the tier is testable * without the real ONNX model. */ -import { describe, it, after, afterEach } from "node:test"; +import { describe, it, after, afterEach, test } from "node:test"; import assert from "node:assert/strict"; import { applyCompressionAsync } from "../../../open-sse/services/compression/index.ts"; @@ -101,3 +101,16 @@ describe("ultra SLM tier — modelPath routes through llmlingua", () => { assert.ok(!techniques(result.stats).includes("ultra-slm")); }); }); + +// ─── Phase 4 (B): SLM-tier resolver, probe, telemetry, pre-warm ────────────── +// (Appended to the pre-existing legacy `modelPath` suite above, which must stay green.) + +import { DEFAULT_COMPRESSION_CONFIG } from "../../../open-sse/services/compression/types.ts"; + +test("DEFAULT_COMPRESSION_CONFIG defaults ultraEngine to 'heuristic'", () => { + assert.equal(DEFAULT_COMPRESSION_CONFIG.ultraEngine, "heuristic"); +}); + +test("DEFAULT_COMPRESSION_CONFIG defaults ultraSlmPrewarm to false", () => { + assert.equal(DEFAULT_COMPRESSION_CONFIG.ultraSlmPrewarm, false); +}); From c9ae9a80ecf4706b1ab804638b476e39a0b74e4f Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:47:28 -0300 Subject: [PATCH 02/18] feat(compression): add CompressionStats.ultraTier resolved-tier signal (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- open-sse/services/compression/types.ts | 8 ++++++++ tests/unit/compression/ultra-slm-tier.test.ts | 15 +++++++++++++++ 2 files changed, 23 insertions(+) diff --git a/open-sse/services/compression/types.ts b/open-sse/services/compression/types.ts index 24b4ac8416e..f02601e574c 100644 --- a/open-sse/services/compression/types.ts +++ b/open-sse/services/compression/types.ts @@ -189,6 +189,14 @@ export interface CompressionStats { validationWarnings?: string[]; validationErrors?: string[]; fallbackApplied?: boolean; + /** + * Phase 4 (B): which `ultra` tier actually ran for this request. + * "slm" — Tier-B ran and produced the output. + * "heuristic-fallback" — Tier-B was selected but failed/timed out → Tier-A used. + * "heuristic" — Tier-A used directly (ultraEngine !== "slm" or SLM unavailable). + * Consumed by D0's persister as `CompressionRunTelemetry.ultraTier`. + */ + ultraTier?: "slm" | "heuristic-fallback" | "heuristic"; preservedBlockCount?: number; rtkRawOutputPointers?: Array<{ id: string; diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index bea1503362a..e06fdfd14c2 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -114,3 +114,18 @@ test("DEFAULT_COMPRESSION_CONFIG defaults ultraEngine to 'heuristic'", () => { test("DEFAULT_COMPRESSION_CONFIG defaults ultraSlmPrewarm to false", () => { assert.equal(DEFAULT_COMPRESSION_CONFIG.ultraSlmPrewarm, false); }); + +import type { CompressionStats } from "../../../open-sse/services/compression/types.ts"; + +test("CompressionStats accepts an optional ultraTier signal", () => { + const s = { + originalTokens: 10, + compressedTokens: 5, + savingsPercent: 50, + techniquesUsed: ["ultra-heuristic-pruning"], + mode: "ultra" as const, + timestamp: 1, + ultraTier: "heuristic" as const, + } satisfies CompressionStats; + assert.equal(s.ultraTier, "heuristic"); +}); From b4ca033b1ebfcf58c4e925e3e62b2ca0cf9216d4 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:48:23 -0300 Subject: [PATCH 03/18] =?UTF-8?q?feat(compression):=20add=20llmlingua=20ul?= =?UTF-8?q?tra=20entry=20=E2=80=94=20slmAvailable=20+=20runLlmlinguaUltra?= =?UTF-8?q?=20+=20prewarm=20(B)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 4.8 (1M context) --- .../engines/llmlingua/ultraEntry.ts | 86 +++++++++++++++++++ .../compression/llmlingua-ultra-entry.test.ts | 44 ++++++++++ 2 files changed, 130 insertions(+) create mode 100644 open-sse/services/compression/engines/llmlingua/ultraEntry.ts create mode 100644 tests/unit/compression/llmlingua-ultra-entry.test.ts diff --git a/open-sse/services/compression/engines/llmlingua/ultraEntry.ts b/open-sse/services/compression/engines/llmlingua/ultraEntry.ts new file mode 100644 index 00000000000..114a84cd770 --- /dev/null +++ b/open-sse/services/compression/engines/llmlingua/ultraEntry.ts @@ -0,0 +1,86 @@ +/** + * LLMLingua-2 entry point for the `ultra` mode (Phase 4, Sub-project B). + * + * A THIN wrapper over the existing worker backend (`./worker.ts`) — no new ONNX + * integration. It adds exactly what the `ultra` two-tier resolver needs: + * - `slmAvailable()` — cached, NON-BLOCKING probe (reuses the worker's memoized + * optional-deps gate). It NEVER loads a model; an actual load happens lazily + * inside the worker under its first-call timeout. + * - `runLlmlinguaUltra(text, opts)` — compress ONE prose string. Throws when the + * backend fail-opens to the original text (no-op), so the ultra resolver can + * fall through to the Tier-A heuristic and record "heuristic-fallback". + * - `prewarmLlmlinguaUltra()` — best-effort warm call (errors swallowed). + * + * The structure-preservation split (code/math/URLs never reach the model) is done + * by the CALLER (`ultra.ts`), exactly as the heuristic path already does — this + * entry only sees prose. + */ + +import { workerBackend, depsAvailable } from "./worker.ts"; +import { DEFAULT_LLMLINGUA_MODEL } from "./constants.ts"; + +/** Cached probe result. null = not probed yet. */ +let _slmAvailable: boolean | null = null; + +/** + * Cheap, cached, non-blocking probe: are the optional SLM deps installed? + * Reuses the worker's memoized `depsAvailable()` (a filesystem manifest check), + * so it never spawns a worker or loads a model. + */ +export function slmAvailable(): boolean { + if (_slmAvailable !== null) return _slmAvailable; + _slmAvailable = depsAvailable(); + return _slmAvailable; +} + +/** Options the ultra SLM tier threads to the worker backend. */ +export interface UltraSlmOptions { + model?: string; + compressionRate?: number; + modelPath?: string; +} + +/** + * Compress ONE prose string via the SLM worker backend. + * + * The worker backend is strictly fail-open: on missing deps / spawn error / + * model-load or inference error / per-call timeout it returns the ORIGINAL text. + * We treat a returned no-op (output not shorter than input) as a FAILURE and + * throw, so the ultra resolver falls back to Tier-A and records the fallback. + */ +export async function runLlmlinguaUltra(text: string, opts?: UltraSlmOptions): Promise { + const out = await workerBackend(text, { + model: opts?.model, + compressionRate: opts?.compressionRate, + modelPath: opts?.modelPath, + }); + if (typeof out !== "string" || out.length >= text.length) { + // Fail-open / no-op from the worker → let the caller fall back to heuristic. + throw new Error("llmlingua-ultra: backend produced no gain"); + } + return out; +} + +/** + * Best-effort pre-warm: ask the worker to load the model once on a short prose + * sample. NEVER throws — any failure is swallowed (the lazy first-call path still + * applies on the next real request). Returns true if a warm call was attempted. + */ +export async function prewarmLlmlinguaUltra(opts?: UltraSlmOptions): Promise { + if (!slmAvailable()) return false; + try { + // A small but non-trivial sample so the worker triggers a real model load. + await workerBackend( + "The quick brown fox jumps over the lazy dog while the sun sets behind the hills.", + { model: opts?.model ?? DEFAULT_LLMLINGUA_MODEL, compressionRate: opts?.compressionRate } + ); + } catch { + // swallow — pre-warm is best-effort + } + return true; +} + +/** Test-only: reset the cached probe. */ +export function __resetUltraEntryForTests(): void { + _slmAvailable = null; +} diff --git a/tests/unit/compression/llmlingua-ultra-entry.test.ts b/tests/unit/compression/llmlingua-ultra-entry.test.ts new file mode 100644 index 00000000000..4238b9494ae --- /dev/null +++ b/tests/unit/compression/llmlingua-ultra-entry.test.ts @@ -0,0 +1,44 @@ +import { test, after } from "node:test"; +import assert from "node:assert/strict"; +import { createRequire } from "node:module"; +import { + slmAvailable, + __resetUltraEntryForTests, +} from "../../../open-sse/services/compression/engines/llmlingua/ultraEntry.ts"; +import { __resetLlmlinguaWorkerForTests } from "../../../open-sse/services/compression/engines/llmlingua/worker.ts"; + +const require = createRequire(import.meta.url); + +function depsResolve(): boolean { + try { + require.resolve("@atjsh/llmlingua-2"); + return true; + } catch { + return false; + } +} + +after(() => { + __resetUltraEntryForTests(); + __resetLlmlinguaWorkerForTests(); +}); + +test("slmAvailable() is false and fast when optional deps are absent", () => { + if (depsResolve()) { + console.log("skip: optional deps present — absent-probe test N/A"); + return; + } + const start = Date.now(); + const available = slmAvailable(); + const elapsed = Date.now() - start; + assert.equal(available, false); + assert.ok(elapsed < 1000, `expected <1000ms, got ${elapsed}ms`); +}); + +test("slmAvailable() result is cached (second call also fast)", () => { + if (depsResolve()) return; + const start = Date.now(); + slmAvailable(); + slmAvailable(); + assert.ok(Date.now() - start < 1000); +}); From 284802d8f2d5f5d43386b76f629fa394f3b1baf0 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:48:56 -0300 Subject: [PATCH 04/18] test(compression): runLlmlinguaUltra throws on backend no-op (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- .../unit/compression/llmlingua-ultra-entry.test.ts | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/tests/unit/compression/llmlingua-ultra-entry.test.ts b/tests/unit/compression/llmlingua-ultra-entry.test.ts index 4238b9494ae..bbb093b75a5 100644 --- a/tests/unit/compression/llmlingua-ultra-entry.test.ts +++ b/tests/unit/compression/llmlingua-ultra-entry.test.ts @@ -42,3 +42,17 @@ test("slmAvailable() result is cached (second call also fast)", () => { slmAvailable(); assert.ok(Date.now() - start < 1000); }); + +import { runLlmlinguaUltra } from "../../../open-sse/services/compression/engines/llmlingua/ultraEntry.ts"; + +test("runLlmlinguaUltra throws when the backend fail-opens (no gain)", async () => { + if (depsResolve()) { + console.log("skip: optional deps present — no-op path N/A"); + return; + } + // Deps absent → workerBackend returns the original text unchanged (no-op) → throw. + await assert.rejects( + () => runLlmlinguaUltra("hello world this is some prose to compress"), + /no gain/ + ); +}); From 5f227c4216aaa97fee5498d7248284967177753f Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:50:42 -0300 Subject: [PATCH 05/18] =?UTF-8?q?feat(compression):=20ultra=20two-tier=20r?= =?UTF-8?q?esolver=20=E2=80=94=20async=20ultraCompress=20+=20sync=20ultraC?= =?UTF-8?q?ompressHeuristic=20(B)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 4.8 (1M context) --- open-sse/services/compression/ultra.ts | 173 +++++++++++++++++- tests/unit/compression/ultra-slm-tier.test.ts | 33 ++++ 2 files changed, 204 insertions(+), 2 deletions(-) diff --git a/open-sse/services/compression/ultra.ts b/open-sse/services/compression/ultra.ts index af02057a2f7..20bd85eff0e 100644 --- a/open-sse/services/compression/ultra.ts +++ b/open-sse/services/compression/ultra.ts @@ -3,9 +3,37 @@ import { extractPreservedBlocks } from "./preservation.ts"; import { DEFAULT_ULTRA_CONFIG } from "./types.ts"; import type { UltraConfig, CompressionStats, CompressionMode } from "./types.ts"; import { extractTextContent, mapTextContent, type ChatMessageLike } from "./messageContent.ts"; +import { slmAvailable, runLlmlinguaUltra } from "./engines/llmlingua/ultraEntry.ts"; const COMPRESSED_PREFIX = "[COMPRESSED:"; +/** + * Async sibling of `mapTextContent`: applies an async transform to each text part + * of a message's content (string content → single call; array content → each + * `{type:"text"}` part). Non-text parts and structure are preserved exactly. + */ +async function mapTextContentAsync( + msg: Message, + fn: (text: string) => Promise +): Promise { + if (typeof msg.content === "string") { + return { ...msg, content: await fn(msg.content) }; + } + if (Array.isArray(msg.content)) { + const next: unknown[] = []; + for (const part of msg.content) { + const p = part as Record; + if (p && p["type"] === "text" && typeof p["text"] === "string") { + next.push({ ...p, text: await fn(p["text"] as string) }); + } else { + next.push(part); + } + } + return { ...msg, content: next }; + } + return msg; +} + /** * Prune PROSE only. Fenced code, inline code, URLs, CONST_CASE, versions, etc. are * tombstoned by `extractPreservedBlocks` and re-stitched verbatim, so the heuristic @@ -34,6 +62,51 @@ function pruneProseOnly(text: string, rate: number, minScore: number): string { .join(""); } +/** + * Compress one prose string with the SLM, preserving code/math/URLs verbatim. + * Reuses `extractPreservedBlocks` (same tombstoning as `pruneProseOnly`), sends + * ONLY prose to the worker backend, and re-stitches preserved blocks unchanged. + * Any backend failure (throw / no-op) falls back to the Tier-A heuristic for that + * segment, so the SLM NEVER touches structured content and NEVER fails the segment. + */ +async function compressProseSlm(text: string, cfg: UltraConfig): Promise { + const { text: withPlaceholders, blocks } = extractPreservedBlocks(text); + if (blocks.length === 0) { + return slmOrHeuristic(text, cfg); + } + const placeholderToContent = new Map(blocks.map((b) => [b.placeholder, b.content])); + const escaped = blocks.map((b) => b.placeholder.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")); + const splitRe = new RegExp(`(${escaped.join("|")})`, "g"); + const parts = withPlaceholders.split(splitRe); + const out: string[] = []; + for (const part of parts) { + if (!part) { + out.push(""); + continue; + } + const preserved = placeholderToContent.get(part); + if (preserved !== undefined) { + out.push(preserved); // verbatim — never sent to the model + } else { + out.push(await slmOrHeuristic(part, cfg)); + } + } + return out.join(""); +} + +/** Run the SLM on a prose segment; on throw/no-op, fall back to the Tier-A pruner for it. */ +async function slmOrHeuristic(prose: string, cfg: UltraConfig): Promise { + try { + return await runLlmlinguaUltra(prose, { + model: cfg.modelPath ? undefined : undefined, + compressionRate: cfg.compressionRate, + modelPath: cfg.modelPath, + }); + } catch { + return pruneByScore(prose, cfg.compressionRate, cfg.minScoreThreshold); + } +} + export interface UltraCompressResult { messages: Array<{ role: string; content?: string | unknown[]; [key: string]: unknown }>; stats: CompressionStats; @@ -41,9 +114,19 @@ export interface UltraCompressResult { type Message = ChatMessageLike; -export function ultraCompress( +/** Tier the ultra resolver records on the stats. */ +export type UltraTier = "slm" | "heuristic-fallback" | "heuristic"; + +/** + * Tier-A heuristic ultra (PURE, SYNCHRONOUS). Identical to the pre-B behaviour. + * Used directly by the stacked sync engine (`cavemanAdapter`) and as the fallback + * tier inside `ultraCompress`. `tier` lets the async resolver tag the resolved + * tier as either "heuristic" (chosen directly) or "heuristic-fallback" (SLM failed). + */ +export function ultraCompressHeuristic( messages: Message[], - config: Partial = {} + config: Partial = {}, + tier: UltraTier = "heuristic" ): UltraCompressResult { const start = Date.now(); const effectiveConfig: UltraConfig = { @@ -93,7 +176,93 @@ export function ultraCompress( mode: "ultra" as CompressionMode, timestamp: Date.now(), durationMs: Date.now() - start, + ultraTier: tier, }; return { messages: compressed, stats }; } + +/** + * Ultra compression with the two-tier resolver (Phase 4, Sub-project B). + * + * - `ultraEngine: "slm"` AND `slmAvailable()` → route prose through the SLM worker + * backend (Tier-B). On timeout / worker error / load failure / no-op → fall back + * to the Tier-A heuristic for THIS request and record "heuristic-fallback". + * - otherwise → Tier-A heuristic ("heuristic"). + * + * The structure-preservation wrapper (`extractPreservedBlocks` / re-stitch, inside + * `pruneProseOnly` for Tier-A and `splitProseAndPreserved` inside the worker engine + * path for Tier-B) wraps BOTH tiers, so code/math/URLs stay verbatim regardless. + * A request is NEVER failed or left uncompressed because the SLM was unavailable. + */ +export async function ultraCompress( + messages: Message[], + config: Partial & { ultraEngine?: "heuristic" | "slm" } = {} +): Promise { + if (config.ultraEngine !== "slm" || !slmAvailable()) { + return ultraCompressHeuristic(messages, config, "heuristic"); + } + + const start = Date.now(); + const effectiveConfig: UltraConfig = { ...DEFAULT_ULTRA_CONFIG, ...config }; + const { maxTokensPerMessage } = effectiveConfig; + + let originalChars = 0; + let compressedChars = 0; + let anySlm = false; + + try { + const compressed: Message[] = []; + for (const msg of messages) { + if (effectiveConfig.preserveSystemPrompt !== false && msg.role === "system") { + compressed.push(msg); + continue; + } + const text = extractTextContent(msg.content); + if (!text || text.startsWith(COMPRESSED_PREFIX)) { + compressed.push(msg); + continue; + } + if (maxTokensPerMessage > 0 && Math.ceil(text.length / 4) <= maxTokensPerMessage) { + compressed.push(msg); + continue; + } + + let messageOriginalChars = 0; + let messageCompressedChars = 0; + const next = (await mapTextContentAsync(msg, async (textPart) => { + if (!textPart || textPart.startsWith(COMPRESSED_PREFIX)) return textPart; + messageOriginalChars += textPart.length; + const out = await compressProseSlm(textPart, effectiveConfig); + if (out !== textPart) anySlm = true; + messageCompressedChars += out.length; + return out; + })) as Message; + originalChars += messageOriginalChars; + compressedChars += messageCompressedChars; + compressed.push(next); + } + + const originalTokens = Math.ceil(originalChars / 4); + const compressedTokens = Math.ceil(compressedChars / 4); + const savingsPercent = + originalTokens > 0 + ? Math.round(((originalTokens - compressedTokens) / originalTokens) * 100 * 10) / 10 + : 0; + + const stats: CompressionStats = { + originalTokens, + compressedTokens, + savingsPercent, + techniquesUsed: anySlm ? ["ultra-slm"] : ["ultra-heuristic-pruning"], + mode: "ultra" as CompressionMode, + timestamp: Date.now(), + durationMs: Date.now() - start, + ultraTier: anySlm ? "slm" : "heuristic-fallback", + }; + return { messages: compressed, stats }; + } catch { + // Any unexpected error in the SLM path → whole-request fail-open to Tier-A. + return ultraCompressHeuristic(messages, config, "heuristic-fallback"); + } +} diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index e06fdfd14c2..b07a90d6b01 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -129,3 +129,36 @@ test("CompressionStats accepts an optional ultraTier signal", () => { } satisfies CompressionStats; assert.equal(s.ultraTier, "heuristic"); }); + +import { + ultraCompress, + ultraCompressHeuristic, +} from "../../../open-sse/services/compression/ultra.ts"; + +test("ultraCompressHeuristic is a synchronous pure heuristic (no SLM)", () => { + const cfg = { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + }; + const r = ultraCompressHeuristic( + [{ role: "user", content: "the quick brown fox jumps over the lazy dog" }], + cfg + ); + assert.equal(r.stats.mode, "ultra"); + assert.equal(r.stats.ultraTier, "heuristic"); + assert.ok(r.stats.techniquesUsed.includes("ultra-heuristic-pruning")); +}); + +test("ultraCompress with default config (no ultraEngine) uses heuristic tier", async () => { + const r = await ultraCompress([{ role: "user", content: "the quick brown fox jumps" }], { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + }); + assert.equal(r.stats.ultraTier, "heuristic"); +}); From f67ca7fea3ffaba90bc876c0a546096e19e5a5e5 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:52:53 -0300 Subject: [PATCH 06/18] feat(compression): ultra SLM tier resolver branches + injectable test seam (B) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The resolved tier is now derived from whether the SLM backend itself produced the segment (ProseSlmResult.usedSlm), not from text inequality — a heuristic-fallback also shrinks the text, so deriving the tier from inequality mislabeled a fallback as 'slm'. Fixes the plan's own 'backend throws -> heuristic-fallback' assertion. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../engines/llmlingua/ultraEntry.ts | 23 +++++- open-sse/services/compression/ultra.ts | 37 ++++++--- tests/unit/compression/ultra-slm-tier.test.ts | 77 +++++++++++++++++++ 3 files changed, 127 insertions(+), 10 deletions(-) diff --git a/open-sse/services/compression/engines/llmlingua/ultraEntry.ts b/open-sse/services/compression/engines/llmlingua/ultraEntry.ts index 114a84cd770..86b621e3a20 100644 --- a/open-sse/services/compression/engines/llmlingua/ultraEntry.ts +++ b/open-sse/services/compression/engines/llmlingua/ultraEntry.ts @@ -22,12 +22,25 @@ import { DEFAULT_LLMLINGUA_MODEL } from "./constants.ts"; /** Cached probe result. null = not probed yet. */ let _slmAvailable: boolean | null = null; +// ─── test-only injectable hooks ───────────────────────────────────────────── +interface UltraSlmTestHooks { + available?: boolean; + run?: (text: string, opts?: UltraSlmOptions) => Promise; +} +let _testHooks: UltraSlmTestHooks | null = null; + +/** Test-only: override availability + the per-prose run, to avoid loading a real model. */ +export function __setUltraSlmTestHooks(hooks: UltraSlmTestHooks): void { + _testHooks = hooks; +} + /** * Cheap, cached, non-blocking probe: are the optional SLM deps installed? * Reuses the worker's memoized `depsAvailable()` (a filesystem manifest check), * so it never spawns a worker or loads a model. */ export function slmAvailable(): boolean { + if (_testHooks && typeof _testHooks.available === "boolean") return _testHooks.available; if (_slmAvailable !== null) return _slmAvailable; _slmAvailable = depsAvailable(); return _slmAvailable; @@ -49,6 +62,13 @@ export interface UltraSlmOptions { * throw, so the ultra resolver falls back to Tier-A and records the fallback. */ export async function runLlmlinguaUltra(text: string, opts?: UltraSlmOptions): Promise { + if (_testHooks?.run) { + const out = await _testHooks.run(text, opts); + if (typeof out !== "string" || out.length >= text.length) { + throw new Error("llmlingua-ultra: backend produced no gain"); + } + return out; + } const out = await workerBackend(text, { model: opts?.model, compressionRate: opts?.compressionRate, @@ -80,7 +100,8 @@ export async function prewarmLlmlinguaUltra(opts?: UltraSlmOptions): Promise { +/** + * One compressed prose segment plus whether the SLM (not the heuristic fallback) + * genuinely produced it. `usedSlm` is what the resolver records the tier from — it + * must reflect "Tier-B ran", NOT merely "the text changed" (a heuristic-fallback + * also shrinks the text, so deriving the tier from text inequality would mislabel + * a fallback as "slm"). + */ +interface ProseSlmResult { + text: string; + usedSlm: boolean; +} + +async function compressProseSlm(text: string, cfg: UltraConfig): Promise { const { text: withPlaceholders, blocks } = extractPreservedBlocks(text); if (blocks.length === 0) { return slmOrHeuristic(text, cfg); @@ -79,6 +91,7 @@ async function compressProseSlm(text: string, cfg: UltraConfig): Promise const splitRe = new RegExp(`(${escaped.join("|")})`, "g"); const parts = withPlaceholders.split(splitRe); const out: string[] = []; + let usedSlm = false; for (const part of parts) { if (!part) { out.push(""); @@ -88,22 +101,28 @@ async function compressProseSlm(text: string, cfg: UltraConfig): Promise if (preserved !== undefined) { out.push(preserved); // verbatim — never sent to the model } else { - out.push(await slmOrHeuristic(part, cfg)); + const seg = await slmOrHeuristic(part, cfg); + out.push(seg.text); + if (seg.usedSlm) usedSlm = true; } } - return out.join(""); + return { text: out.join(""), usedSlm }; } -/** Run the SLM on a prose segment; on throw/no-op, fall back to the Tier-A pruner for it. */ -async function slmOrHeuristic(prose: string, cfg: UltraConfig): Promise { +/** + * Run the SLM on a prose segment; on throw/no-op, fall back to the Tier-A pruner for it. + * `usedSlm` is true ONLY when the SLM backend itself produced the output. + */ +async function slmOrHeuristic(prose: string, cfg: UltraConfig): Promise { try { - return await runLlmlinguaUltra(prose, { + const text = await runLlmlinguaUltra(prose, { model: cfg.modelPath ? undefined : undefined, compressionRate: cfg.compressionRate, modelPath: cfg.modelPath, }); + return { text, usedSlm: true }; } catch { - return pruneByScore(prose, cfg.compressionRate, cfg.minScoreThreshold); + return { text: pruneByScore(prose, cfg.compressionRate, cfg.minScoreThreshold), usedSlm: false }; } } @@ -233,8 +252,8 @@ export async function ultraCompress( const next = (await mapTextContentAsync(msg, async (textPart) => { if (!textPart || textPart.startsWith(COMPRESSED_PREFIX)) return textPart; messageOriginalChars += textPart.length; - const out = await compressProseSlm(textPart, effectiveConfig); - if (out !== textPart) anySlm = true; + const { text: out, usedSlm } = await compressProseSlm(textPart, effectiveConfig); + if (usedSlm) anySlm = true; messageCompressedChars += out.length; return out; })) as Message; diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index b07a90d6b01..ff037582720 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -162,3 +162,80 @@ test("ultraCompress with default config (no ultraEngine) uses heuristic tier", a }); assert.equal(r.stats.ultraTier, "heuristic"); }); + +import { + __setUltraSlmTestHooks, + __resetUltraEntryForTests, +} from "../../../open-sse/services/compression/engines/llmlingua/ultraEntry.ts"; + +test("ultraEngine:'slm' with available stub backend records ultraTier:'slm'", async () => { + __setUltraSlmTestHooks({ + available: true, + run: async (text) => text.slice(0, Math.ceil(text.length / 2)), + }); + try { + const r = await ultraCompress( + [{ role: "user", content: "the quick brown fox jumps over the lazy dog repeatedly today" }], + { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + ultraEngine: "slm", + } + ); + assert.equal(r.stats.ultraTier, "slm"); + assert.ok(r.stats.techniquesUsed.includes("ultra-slm")); + assert.ok(r.stats.compressedTokens <= r.stats.originalTokens); + } finally { + __resetUltraEntryForTests(); + } +}); + +test("ultraEngine:'slm' but backend throws → ultraTier:'heuristic-fallback'", async () => { + __setUltraSlmTestHooks({ + available: true, + run: async () => { + throw new Error("worker timeout"); + }, + }); + try { + const r = await ultraCompress( + [{ role: "user", content: "the quick brown fox jumps over the lazy dog repeatedly today" }], + { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + ultraEngine: "slm", + } + ); + assert.equal(r.stats.ultraTier, "heuristic-fallback"); + } finally { + __resetUltraEntryForTests(); + } +}); + +test("ultraEngine:'slm' but slmAvailable() false → heuristic tier (no SLM attempt)", async () => { + __setUltraSlmTestHooks({ + available: false, + run: async () => { + throw new Error("should not be called"); + }, + }); + try { + const r = await ultraCompress([{ role: "user", content: "the quick brown fox jumps" }], { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + ultraEngine: "slm", + }); + assert.equal(r.stats.ultraTier, "heuristic"); + } finally { + __resetUltraEntryForTests(); + } +}); From 192645c60bf51eeccb3ef790ea57f8ced9f9ab18 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:55:15 -0300 Subject: [PATCH 07/18] test(compression): ultra SLM tier preserves code/URL structure (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- tests/unit/compression/ultra-slm-tier.test.ts | 29 +++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index ff037582720..e792d48a77c 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -239,3 +239,32 @@ test("ultraEngine:'slm' but slmAvailable() false → heuristic tier (no SLM atte __resetUltraEntryForTests(); } }); + +test("SLM tier preserves fenced code + URLs verbatim (structure wrapper)", async () => { + // Stub the SLM to lowercase prose — any leakage of code/URL into it would show. + __setUltraSlmTestHooks({ + available: true, + run: async (text) => (text.trim() ? text.toLowerCase() + " x" : text), + }); + try { + const code = "```js\nconst A = 1; // KEEP\n```"; + const url = "https://Example.com/Path"; + // The fenced block must open at line-start for `extractPreservedBlocks` to tombstone + // it (the same rule the heuristic Tier-A already relies on); both tiers share that + // wrapper, so this proves the SLM tier preserves structure identically. + const content = `Some PROSE here and ${url} trailing PROSE\n${code}\nmore PROSE after`; + const r = await ultraCompress([{ role: "user", content }], { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + ultraEngine: "slm", + }); + const out = r.messages[0].content as string; + assert.ok(out.includes(code), "fenced code block must survive verbatim"); + assert.ok(out.includes(url), "URL must survive verbatim"); + } finally { + __resetUltraEntryForTests(); + } +}); From c8552c67ae5f399622697e09472d38f2e6249340 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:56:01 -0300 Subject: [PATCH 08/18] fix(compression): stacked ultra engine uses sync ultraCompressHeuristic (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- .../services/compression/engines/cavemanAdapter.ts | 4 ++-- tests/unit/compression/ultra-slm-tier.test.ts | 12 ++++++++++++ 2 files changed, 14 insertions(+), 2 deletions(-) diff --git a/open-sse/services/compression/engines/cavemanAdapter.ts b/open-sse/services/compression/engines/cavemanAdapter.ts index 41ea89dc2bc..d819357c0f8 100644 --- a/open-sse/services/compression/engines/cavemanAdapter.ts +++ b/open-sse/services/compression/engines/cavemanAdapter.ts @@ -1,7 +1,7 @@ import { applyLiteCompression } from "../lite.ts"; import { cavemanCompress } from "../caveman.ts"; import { compressAggressive } from "../aggressive.ts"; -import { ultraCompress } from "../ultra.ts"; +import { ultraCompressHeuristic } from "../ultra.ts"; import { createCompressionStats } from "../stats.ts"; import { adaptBodyForCompression } from "../bodyAdapter.ts"; import { @@ -414,7 +414,7 @@ export const ultraEngine: CompressionEngine = { ...(options?.stepConfig ?? {}), preserveSystemPrompt: options?.config?.preserveSystemPrompt !== false, }; - const result = ultraCompress(messages, ultraConfig); + const result = ultraCompressHeuristic(messages, ultraConfig); const compressedBody = { ...adapter.body, messages: result.messages }; return { body: adapter.restore(compressedBody), diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index e792d48a77c..00a50365e0c 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -268,3 +268,15 @@ test("SLM tier preserves fenced code + URLs verbatim (structure wrapper)", async __resetUltraEntryForTests(); } }); + +import { ultraEngine } from "../../../open-sse/services/compression/engines/cavemanAdapter.ts"; + +test("stacked ultraEngine.apply stays synchronous and compresses via heuristic", () => { + const res = ultraEngine.apply( + { messages: [{ role: "user", content: "the quick brown fox jumps over the lazy dog" }] }, + { config: { ultra: { compressionRate: 0.5 } } as never } + ); + // Synchronous result object (not a Promise), with a real stats record. + assert.equal(typeof (res as { then?: unknown }).then, "undefined"); + assert.ok(res.stats); +}); From 14c740b43c32ff1af047e3b35b42f1a8a575f53a Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:57:23 -0300 Subject: [PATCH 09/18] feat(compression): route ultra SLM tier through async ultraCompress in strategySelector (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- .../services/compression/strategySelector.ts | 59 +++++++++++++++---- tests/unit/compression/ultra-slm-tier.test.ts | 31 ++++++++++ 2 files changed, 78 insertions(+), 12 deletions(-) diff --git a/open-sse/services/compression/strategySelector.ts b/open-sse/services/compression/strategySelector.ts index 51e5865122b..41fcb82b50a 100644 --- a/open-sse/services/compression/strategySelector.ts +++ b/open-sse/services/compression/strategySelector.ts @@ -9,7 +9,7 @@ import type { CompressionEngineApplyOptions } from "./engines/types.ts"; import { applyLiteCompression } from "./lite.ts"; import { cavemanCompress } from "./caveman.ts"; import { compressAggressive } from "./aggressive.ts"; -import { ultraCompress } from "./ultra.ts"; +import { ultraCompress, ultraCompressHeuristic } from "./ultra.ts"; import { createCompressionStats } from "./stats.ts"; import { registerBuiltinCompressionEngines } from "./engines/index.ts"; import { getCompressionEngine, getEngineEntry } from "./engines/registry.ts"; @@ -325,19 +325,22 @@ export function applyCompression( ...(options?.config?.ultra ?? {}), preserveSystemPrompt: options?.config?.preserveSystemPrompt !== false, }; - const result = ultraCompress(messages, ultraConfig); + const result = ultraCompressHeuristic(messages, ultraConfig); const compressedBody = { ...compressionBody, messages: result.messages }; return { body: adapter.restore(compressedBody), compressed: result.stats.savingsPercent > 0, - stats: createCompressionStats( - compressionBody, - compressedBody, - mode, - ["ultra"], - result.stats.rulesApplied, - result.stats.durationMs - ), + stats: { + ...createCompressionStats( + compressionBody, + compressedBody, + mode, + ["ultra"], + result.stats.rulesApplied, + result.stats.durationMs + ), + ultraTier: result.stats.ultraTier, + }, }; } return { body, compressed: false, stats: null }; @@ -402,9 +405,41 @@ async function applyUltraAsync( const ultraConfig = options?.config?.ultra; const modelPath = typeof ultraConfig?.modelPath === "string" ? ultraConfig.modelPath.trim() : ""; - // No model configured → heuristic ultra (unchanged default). + // No explicit modelPath → run the two-tier ultra resolver (heuristic, or SLM when + // config.ultraEngine === "slm" and the worker backend is available). This is the + // Phase-4 (B) path; it fail-opens to the heuristic and records the resolved tier. if (!modelPath) { - return applyCompression(body, "ultra", options); + const adapter = adaptBodyForCompression(body); + const messages = (adapter.body.messages ?? []) as Array<{ + role: string; + content?: string | unknown[]; + [key: string]: unknown; + }>; + if (!Array.isArray(messages) || messages.length === 0) { + return { body, compressed: false, stats: null }; + } + const ultraConfig = { + ...(options?.config?.ultra ?? {}), + preserveSystemPrompt: options?.config?.preserveSystemPrompt !== false, + ultraEngine: options?.config?.ultraEngine, + }; + const result = await ultraCompress(messages, ultraConfig); + const compressedBody = { ...adapter.body, messages: result.messages }; + return { + body: adapter.restore(compressedBody), + compressed: result.stats.savingsPercent > 0, + stats: { + ...createCompressionStats( + adapter.body, + compressedBody, + "ultra", + result.stats.techniquesUsed, + result.stats.rulesApplied, + result.stats.durationMs + ), + ultraTier: result.stats.ultraTier, + }, + }; } registerBuiltinCompressionEngines(); diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index 00a50365e0c..b043b0b75be 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -280,3 +280,34 @@ test("stacked ultraEngine.apply stays synchronous and compresses via heuristic", assert.equal(typeof (res as { then?: unknown }).then, "undefined"); assert.ok(res.stats); }); + +test("applyCompressionAsync ultra + ultraEngine:'slm' (stub) yields ultraTier in stats", async () => { + __setUltraSlmTestHooks({ + available: true, + run: async (text) => text.slice(0, Math.ceil(text.length / 2)), + }); + try { + const reqBody = { + messages: [ + { role: "user", content: "the quick brown fox jumps over the lazy dog more than once" }, + ], + }; + const result = await applyCompressionAsync(reqBody, "ultra", { + config: { + enabled: true, + defaultMode: "ultra", + ultraEngine: "slm", + ultra: { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + }, + } as never, + }); + assert.equal(result.stats?.ultraTier, "slm"); + } finally { + __resetUltraEntryForTests(); + } +}); From 7b8427ba63d016320b7cb1e71177e8aec23a309a Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:58:07 -0300 Subject: [PATCH 10/18] feat(compression): re-export ultra-SLM surface from compression index (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- open-sse/services/compression/index.ts | 7 +++++++ tests/unit/compression/ultra-slm-tier.test.ts | 9 +++++++++ 2 files changed, 16 insertions(+) diff --git a/open-sse/services/compression/index.ts b/open-sse/services/compression/index.ts index 062a1c8d57b..8ef9a581b57 100644 --- a/open-sse/services/compression/index.ts +++ b/open-sse/services/compression/index.ts @@ -153,6 +153,13 @@ export { STOPWORDS, FORCE_PRESERVE_RE, scoreToken, pruneByScore } from "./ultraH export type { UltraCompressResult } from "./ultra.ts"; export { ultraCompress } from "./ultra.ts"; +export { ultraCompressHeuristic } from "./ultra.ts"; +export type { UltraTier } from "./ultra.ts"; +export { + slmAvailable, + runLlmlinguaUltra, + prewarmLlmlinguaUltra, +} from "./engines/llmlingua/ultraEntry.ts"; export type { UltraConfig } from "./types.ts"; export { DEFAULT_ULTRA_CONFIG } from "./types.ts"; diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index b043b0b75be..fe20472ed38 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -311,3 +311,12 @@ test("applyCompressionAsync ultra + ultraEngine:'slm' (stub) yields ultraTier in __resetUltraEntryForTests(); } }); + +import * as compression from "../../../open-sse/services/compression/index.ts"; + +test("compression index re-exports the ultra-SLM surface", () => { + assert.equal(typeof compression.ultraCompressHeuristic, "function"); + assert.equal(typeof compression.slmAvailable, "function"); + assert.equal(typeof compression.runLlmlinguaUltra, "function"); + assert.equal(typeof compression.prewarmLlmlinguaUltra, "function"); +}); From 74eebbc1e34d39d536f826d3a9394a7351571884 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:58:54 -0300 Subject: [PATCH 11/18] feat(compression): ultra SLM pre-warm fires one best-effort warm call (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- .../engines/llmlingua/ultraEntry.ts | 6 ++- .../compression/llmlingua-ultra-entry.test.ts | 48 +++++++++++++++++++ 2 files changed, 52 insertions(+), 2 deletions(-) diff --git a/open-sse/services/compression/engines/llmlingua/ultraEntry.ts b/open-sse/services/compression/engines/llmlingua/ultraEntry.ts index 86b621e3a20..58e4107fefc 100644 --- a/open-sse/services/compression/engines/llmlingua/ultraEntry.ts +++ b/open-sse/services/compression/engines/llmlingua/ultraEntry.ts @@ -90,12 +90,14 @@ export async function prewarmLlmlinguaUltra(opts?: UltraSlmOptions): Promise { + let calls = 0; + __setUltraSlmTestHooks({ + available: true, + run: async (text) => { + calls++; + return text.slice(0, 1); + }, + }); + try { + const attempted = await prewarmLlmlinguaUltra(); + assert.equal(attempted, true); + assert.equal(calls, 1); + } finally { + __resetUltraEntryForTests(); + } +}); + +test("prewarmLlmlinguaUltra swallows a warm-call failure", async () => { + __setUltraSlmTestHooks({ + available: true, + run: async () => { + throw new Error("load failed"); + }, + }); + try { + const attempted = await prewarmLlmlinguaUltra(); // must NOT throw + assert.equal(attempted, true); + } finally { + __resetUltraEntryForTests(); + } +}); + +test("prewarmLlmlinguaUltra is a no-op when unavailable", async () => { + __setUltraSlmTestHooks({ available: false }); + try { + const attempted = await prewarmLlmlinguaUltra(); + assert.equal(attempted, false); + } finally { + __resetUltraEntryForTests(); + } +}); From e3866122492e85150d09a034898173e24ae58085 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 18:59:46 -0300 Subject: [PATCH 12/18] feat(compression): pure shouldPrewarmUltraSlm decision helper (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- open-sse/services/compression/ultra.ts | 13 +++++++++++++ tests/unit/compression/ultra-slm-tier.test.ts | 9 +++++++++ 2 files changed, 22 insertions(+) diff --git a/open-sse/services/compression/ultra.ts b/open-sse/services/compression/ultra.ts index 0cc8f2a1a11..a53aa80c192 100644 --- a/open-sse/services/compression/ultra.ts +++ b/open-sse/services/compression/ultra.ts @@ -285,3 +285,16 @@ export async function ultraCompress( return ultraCompressHeuristic(messages, config, "heuristic-fallback"); } } + +/** + * Pure decision: should the ultra SLM model be pre-warmed for this config? + * True only when the SLM tier is selected AND pre-warm is enabled. The CALLER + * decides timing (enable-transition or cold-start) and fires `prewarmLlmlinguaUltra` + * best-effort; this helper stays clock-free / side-effect-free. + */ +export function shouldPrewarmUltraSlm(config: { + ultraEngine?: "heuristic" | "slm"; + ultraSlmPrewarm?: boolean; +}): boolean { + return config.ultraEngine === "slm" && config.ultraSlmPrewarm === true; +} diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index fe20472ed38..2fbb4beccb7 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -320,3 +320,12 @@ test("compression index re-exports the ultra-SLM surface", () => { assert.equal(typeof compression.runLlmlinguaUltra, "function"); assert.equal(typeof compression.prewarmLlmlinguaUltra, "function"); }); + +import { shouldPrewarmUltraSlm } from "../../../open-sse/services/compression/ultra.ts"; + +test("shouldPrewarmUltraSlm: true only when slm + prewarm both on", () => { + assert.equal(shouldPrewarmUltraSlm({ ultraEngine: "slm", ultraSlmPrewarm: true }), true); + assert.equal(shouldPrewarmUltraSlm({ ultraEngine: "slm", ultraSlmPrewarm: false }), false); + assert.equal(shouldPrewarmUltraSlm({ ultraEngine: "heuristic", ultraSlmPrewarm: true }), false); + assert.equal(shouldPrewarmUltraSlm({}), false); +}); From b1e13c980c162cb49f76754614a6e835faa49a27 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 19:04:12 -0300 Subject: [PATCH 13/18] feat(compression): best-effort ultra SLM pre-warm on enable + cold restart (B) Also completes Task 1's DB-read + Zod surface for the two new fields: - getCompressionSettings reads ultraEngine/ultraSlmPrewarm rows (round-trip); - compressionSettingsUpdateSchema accepts them (.strict() would reject otherwise). The prewarm trigger fires fire-and-forget from the save path (enable transition) and once per process from the cold read path, both via maybePrewarmUltraSlmOnConfig. Co-Authored-By: Claude Opus 4.8 (1M context) --- open-sse/services/compression/ultra.ts | 23 +++++++++++- src/lib/db/compression.ts | 35 ++++++++++++++++++- .../validation/compressionConfigSchemas.ts | 2 ++ tests/unit/compression/db.test.ts | 15 ++++++++ tests/unit/compression/ultra-slm-tier.test.ts | 21 +++++++++++ 5 files changed, 94 insertions(+), 2 deletions(-) diff --git a/open-sse/services/compression/ultra.ts b/open-sse/services/compression/ultra.ts index a53aa80c192..aeb7bcc3fa7 100644 --- a/open-sse/services/compression/ultra.ts +++ b/open-sse/services/compression/ultra.ts @@ -3,7 +3,11 @@ import { extractPreservedBlocks } from "./preservation.ts"; import { DEFAULT_ULTRA_CONFIG } from "./types.ts"; import type { UltraConfig, CompressionStats, CompressionMode } from "./types.ts"; import { extractTextContent, mapTextContent, type ChatMessageLike } from "./messageContent.ts"; -import { slmAvailable, runLlmlinguaUltra } from "./engines/llmlingua/ultraEntry.ts"; +import { + slmAvailable, + runLlmlinguaUltra, + prewarmLlmlinguaUltra, +} from "./engines/llmlingua/ultraEntry.ts"; const COMPRESSED_PREFIX = "[COMPRESSED:"; @@ -298,3 +302,20 @@ export function shouldPrewarmUltraSlm(config: { }): boolean { return config.ultraEngine === "slm" && config.ultraSlmPrewarm === true; } + +/** + * Best-effort: when the resolved config selects the SLM tier WITH pre-warm, + * trigger a single warm call. Awaitable for tests; call sites fire-and-forget + * (`void maybePrewarmUltraSlmOnConfig(cfg)`). Never throws. + */ +export async function maybePrewarmUltraSlmOnConfig(config: { + ultraEngine?: "heuristic" | "slm"; + ultraSlmPrewarm?: boolean; +}): Promise { + if (!shouldPrewarmUltraSlm(config)) return; + try { + await prewarmLlmlinguaUltra(); + } catch { + // best-effort + } +} diff --git a/src/lib/db/compression.ts b/src/lib/db/compression.ts index 301f375789f..75f8c18027e 100644 --- a/src/lib/db/compression.ts +++ b/src/lib/db/compression.ts @@ -28,6 +28,7 @@ import { type RtkConfig, type UltraConfig, } from "@omniroute/open-sse/services/compression/types.ts"; +import { maybePrewarmUltraSlmOnConfig } from "@omniroute/open-sse/services/compression/ultra.ts"; const NAMESPACE = "compression"; const COMPRESSION_MODES = new Set([ @@ -48,6 +49,12 @@ let compressionSettingsCache: { dbRef: WeakRef; } | null = null; +// Phase 4 (B): one cold-start SLM pre-warm attempt per process. The save path fires +// on every enable transition; this guard keeps the read path from re-warming on every +// cache miss (the read path runs at most once per 5s, but a cold start should warm once, +// not repeatedly). Best-effort either way (`maybePrewarmUltraSlmOnConfig` never throws). +let _ultraSlmColdPrewarmAttempted = false; + function toRecord(value: unknown): JsonRecord { return value && typeof value === "object" ? (value as JsonRecord) : {}; } @@ -634,6 +641,14 @@ export async function getCompressionSettings(): Promise { config.activeComboId = typeof parsed === "string" && parsed.trim() ? parsed.trim() : null; break; + case "ultraEngine": + // Phase 4 (B): SLM tier selector. Only the two known values; anything else + // falls back to the heuristic default so a malformed row can never enable SLM. + config.ultraEngine = parsed === "slm" ? "slm" : "heuristic"; + break; + case "ultraSlmPrewarm": + config.ultraSlmPrewarm = parsed === true; + break; } } @@ -658,6 +673,17 @@ export async function getCompressionSettings(): Promise { dbRef: new WeakRef(db), }; + // Phase 4 (B): cold-restart pre-warm — when the stored config already selects the SLM + // tier with pre-warm on, warm the model once (best-effort, fire-and-forget, guarded so + // a frequently-hit read path warms at most once per process). Cache hits return above. + if (!_ultraSlmColdPrewarmAttempted) { + _ultraSlmColdPrewarmAttempted = true; + void maybePrewarmUltraSlmOnConfig({ + ultraEngine: config.ultraEngine, + ultraSlmPrewarm: config.ultraSlmPrewarm, + }); + } + return config; } @@ -686,7 +712,14 @@ export async function updateCompressionSettings( backupDbFile("pre-write"); compressionSettingsCache = null; invalidateDbCache(); - return getCompressionSettings(); + const next = await getCompressionSettings(); + // Phase 4 (B): the SAVE path covers the enable transition — if this write turns the + // SLM tier + pre-warm on, warm the model once (best-effort, fire-and-forget). + void maybePrewarmUltraSlmOnConfig({ + ultraEngine: next.ultraEngine, + ultraSlmPrewarm: next.ultraSlmPrewarm, + }); + return next; } function normalizeMcpAccessibilityConfig(value: unknown): McpAccessibilityConfig { diff --git a/src/shared/validation/compressionConfigSchemas.ts b/src/shared/validation/compressionConfigSchemas.ts index c9128a37817..293c11694fd 100644 --- a/src/shared/validation/compressionConfigSchemas.ts +++ b/src/shared/validation/compressionConfigSchemas.ts @@ -205,6 +205,8 @@ export const compressionSettingsUpdateSchema = z engines: z.record(z.string(), engineToggleSchema).optional(), enginesExplicit: z.boolean().optional(), activeComboId: z.string().nullable().optional(), + ultraEngine: z.enum(["heuristic", "slm"]).optional(), + ultraSlmPrewarm: z.boolean().optional(), }) .strict(); diff --git a/tests/unit/compression/db.test.ts b/tests/unit/compression/db.test.ts index 9da322eb97a..9f4731da426 100644 --- a/tests/unit/compression/db.test.ts +++ b/tests/unit/compression/db.test.ts @@ -139,4 +139,19 @@ describe("updateCompressionSettings", () => { assert.equal(settings.ultra?.modelPath, "/tmp/model.onnx"); assert.equal(settings.ultra?.maxTokensPerMessage, 512); }); + + it("round-trips ultraEngine + ultraSlmPrewarm (Phase 4 B), defaulting off", async () => { + const before = await getCompressionSettings(); + assert.equal(before.ultraEngine, "heuristic"); + assert.equal(before.ultraSlmPrewarm, false); + + await updateCompressionSettings({ + ultraEngine: "slm", + ultraSlmPrewarm: true, + } as any); + + const after = await getCompressionSettings(); + assert.equal(after.ultraEngine, "slm"); + assert.equal(after.ultraSlmPrewarm, true); + }); }); diff --git a/tests/unit/compression/ultra-slm-tier.test.ts b/tests/unit/compression/ultra-slm-tier.test.ts index 2fbb4beccb7..918ac6a2f27 100644 --- a/tests/unit/compression/ultra-slm-tier.test.ts +++ b/tests/unit/compression/ultra-slm-tier.test.ts @@ -329,3 +329,24 @@ test("shouldPrewarmUltraSlm: true only when slm + prewarm both on", () => { assert.equal(shouldPrewarmUltraSlm({ ultraEngine: "heuristic", ultraSlmPrewarm: true }), false); assert.equal(shouldPrewarmUltraSlm({}), false); }); + +import { maybePrewarmUltraSlmOnConfig } from "../../../open-sse/services/compression/ultra.ts"; + +test("maybePrewarmUltraSlmOnConfig fires prewarm when slm+prewarm on (stub)", async () => { + let warmed = 0; + __setUltraSlmTestHooks({ + available: true, + run: async (t) => { + warmed++; + return t.slice(0, 1); + }, + }); + try { + await maybePrewarmUltraSlmOnConfig({ ultraEngine: "slm", ultraSlmPrewarm: true }); + assert.equal(warmed, 1); + await maybePrewarmUltraSlmOnConfig({ ultraEngine: "heuristic", ultraSlmPrewarm: true }); + assert.equal(warmed, 1); // unchanged — heuristic does not prewarm + } finally { + __resetUltraEntryForTests(); + } +}); From a6e87bf29432c21464f5011e11d605d47cbd4c88 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 19:06:46 -0300 Subject: [PATCH 14/18] test(compression): await now-async ultraCompress in code-preservation guard (B) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ultraCompress became async in Task 3; this pre-existing regression guard called it synchronously and destructured messages without awaiting. Aligns the test to the async source — assertions unchanged (code/inline/URL survive, prose still compressed). Co-Authored-By: Claude Opus 4.8 (1M context) --- tests/unit/compression/ultra-code-preservation.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/unit/compression/ultra-code-preservation.test.ts b/tests/unit/compression/ultra-code-preservation.test.ts index fb4ff71891f..990867dc364 100644 --- a/tests/unit/compression/ultra-code-preservation.test.ts +++ b/tests/unit/compression/ultra-code-preservation.test.ts @@ -9,7 +9,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { ultraCompress } from "@omniroute/open-sse/services/compression/ultra.ts"; -test("ultraCompress preserves fenced code, inline code, and URLs byte-identical", () => { +test("ultraCompress preserves fenced code, inline code, and URLs byte-identical", async () => { const code = "```ts\nexport function add(a, b) {\n return a + b;\n}\n```"; const inline = "`add(x, y)`"; const url = "https://example.com/api/v1/auth?id=42"; @@ -20,7 +20,7 @@ test("ultraCompress preserves fenced code, inline code, and URLs byte-identical" // Realistic layout: fenced code block sits on its own line (markdown convention). const text = `${filler}\n\n${code}\n\nThen call ${inline} and see ${url} for details.\n\n${filler}`; - const { messages } = ultraCompress([{ role: "user", content: text }], { + const { messages } = await ultraCompress([{ role: "user", content: text }], { maxTokensPerMessage: 0, }); const out = typeof messages[0].content === "string" ? messages[0].content : ""; From 12aad5aa6e31a4b86d0a91515cfaaf662b09f073 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 19:38:04 -0300 Subject: [PATCH 15/18] feat(compression): surface ultra SLM tier + pre-warm in settings panel (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- .../context/settings/CompressionPanel.tsx | 50 ++++++ src/i18n/messages/en.json | 5 + tests/unit/ui/compressionUltraTier.test.tsx | 162 ++++++++++++++++++ 3 files changed, 217 insertions(+) create mode 100644 tests/unit/ui/compressionUltraTier.test.tsx diff --git a/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx b/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx index eb3f7f1669e..56fed47ffc8 100644 --- a/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx +++ b/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx @@ -50,6 +50,12 @@ interface CompressionConfig { activeComboId: string | null; cavemanOutputMode?: CavemanOutputModeConfig; outputStyles?: Array<{ id: string; level: CavemanIntensity }>; + // Phase 4 (B): two-tier `ultra` mode controls. + // ultraEngine "heuristic" = Tier-A token pruner (default, byte-identical to pre-B); + // "slm" = Tier-B LLMLingua-2 ONNX worker when available, else fail-open to Tier-A. + ultraEngine?: "heuristic" | "slm"; + // Best-effort pre-warm of the SLM model on enable / cold restart. Default false. + ultraSlmPrewarm?: boolean; } const CAVEMAN_OUTPUT_LEVELS: CavemanIntensity[] = ["lite", "full", "ultra"]; @@ -62,6 +68,8 @@ const DEFAULT_CONFIG: CompressionConfig = { activeComboId: null, cavemanOutputMode: { enabled: false, intensity: "full", autoClarity: true }, outputStyles: [], + ultraEngine: "heuristic", + ultraSlmPrewarm: false, }; function normalizeEngines(raw: unknown): Record { @@ -354,6 +362,48 @@ export default function CompressionPanel() { })} + {/* Ultra SLM tier — Phase 4 (B): pick the `ultra`-mode engine (heuristic Tier-A + or the opt-in LLMLingua-2 SLM Tier-B) + best-effort pre-warm. */} +
+ + + {config.ultraEngine === "slm" && ( + <> +

{t("compressionUltraSlmHint")}

+ + + )} +
+ {/* mcpAccessibility — writes its own endpoint / separate store */}
diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 592775aa2ad..beb013043f3 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -5595,6 +5595,11 @@ "compressionUltraMinScore": "Minimum Score Threshold", "compressionUltraSlmFallback": "Fallback to Aggressive", "compressionUltraModelPath": "SLM Model Path", + "compressionUltraEngine": "Ultra tier", + "compressionUltraEngineHeuristic": "Heuristic (Tier-A, default)", + "compressionUltraEngineSlm": "SLM (LLMLingua-2, opt-in)", + "compressionUltraSlmHint": "SLM downloads a small ONNX model on first use (cold-start) and transparently falls back to the heuristic on timeout or if unavailable.", + "compressionUltraSlmPrewarm": "Pre-warm SLM model on enable", "compressionSummarizerEnabled": "Enable Summarizer", "compressionMaxTokensPerMessage": "Max Tokens Per Message", "compressionMinSavings": "Min Savings Threshold", diff --git a/tests/unit/ui/compressionUltraTier.test.tsx b/tests/unit/ui/compressionUltraTier.test.tsx new file mode 100644 index 00000000000..00355a80c92 --- /dev/null +++ b/tests/unit/ui/compressionUltraTier.test.tsx @@ -0,0 +1,162 @@ +// @vitest-environment jsdom +import React, { act } from "react"; +import { createRoot } from "react-dom/client"; +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; + +// i18n does not resolve to a real locale in vitest/jsdom, so mock next-intl to echo +// the key. This test asserts ONLY on i18n-independent hooks (data-testid + values) +// and the captured PUT body. +vi.mock("next-intl", () => ({ useTranslations: () => (key: string) => key })); + +const containers: HTMLElement[] = []; +const roots: Array<{ unmount: () => void }> = []; + +function mount(ui: React.ReactElement): HTMLElement { + const container = document.createElement("div"); + document.body.appendChild(container); + containers.push(container); + const root = createRoot(container); + roots.push(root); + act(() => root.render(ui)); + return container; +} + +beforeEach(() => { + (globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = + true; +}); + +afterEach(async () => { + vi.restoreAllMocks(); + await act(async () => { + while (roots.length > 0) roots.pop()?.unmount(); + }); + for (let i = 0; i < 10; i++) await Promise.resolve(); + while (containers.length > 0) containers.pop()?.remove(); + document.body.innerHTML = ""; +}); + +async function flush() { + await act(async () => { + for (let i = 0; i < 10; i++) await Promise.resolve(); + }); +} + +interface CapturedPut { + url: string; + body: Record; +} + +function setupFetchMock(overrides?: Record): { puts: CapturedPut[] } { + const puts: CapturedPut[] = []; + const json = (body: unknown, status = 200) => + new Response(JSON.stringify(body), { status, headers: { "Content-Type": "application/json" } }); + const initial = { + enabled: true, + autoTriggerTokens: 0, + preserveSystemPrompt: true, + engines: {}, + activeComboId: null, + outputStyles: [], + cavemanOutputMode: { enabled: false, intensity: "full", autoClarity: true }, + ultraEngine: "heuristic", + ultraSlmPrewarm: false, + ...overrides, + }; + vi.spyOn(globalThis, "fetch").mockImplementation( + async (input: RequestInfo | URL, init?: RequestInit) => { + const url = input.toString(); + const method = (init?.method ?? "GET").toUpperCase(); + if (url.includes("/api/settings/compression/mcp-accessibility")) return json({ enabled: true }); + if (url.includes("/api/settings/compression")) { + if (method === "PUT") { + const body = JSON.parse(String(init?.body ?? "{}")); + puts.push({ url, body }); + return json({ ...initial, ...body }); + } + return json(initial); + } + return json({}, 404); + } + ); + return { puts }; +} + +describe("CompressionPanel ultra SLM tier", () => { + it("renders the ultra-engine select defaulting to heuristic", async () => { + setupFetchMock(); + const { default: CompressionPanel } = await import( + "../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel" + ); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="ultra-engine-select"]` + ) as HTMLSelectElement | null; + expect(select, "ultra-engine select must render").toBeTruthy(); + expect(select?.value).toBe("heuristic"); + // The pre-warm toggle is hidden while heuristic is selected. + expect( + container.querySelector(`[data-testid="ultra-slm-prewarm-toggle"]`) + ).toBeFalsy(); + }); + + it("selecting SLM PUTs ultraEngine:'slm' and reveals the pre-warm toggle", async () => { + const { puts } = setupFetchMock(); + const { default: CompressionPanel } = await import( + "../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel" + ); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="ultra-engine-select"]` + ) as HTMLSelectElement; + expect(select).toBeTruthy(); + await act(async () => { + select.value = "slm"; + select.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + + const put = puts.find((p) => "ultraEngine" in p.body); + expect(put, "a PUT carrying ultraEngine").toBeTruthy(); + expect(put!.body.ultraEngine).toBe("slm"); + + // Pre-warm toggle now visible. + const toggle = container.querySelector(`[data-testid="ultra-slm-prewarm-toggle"]`); + expect(toggle, "pre-warm toggle must appear when slm is selected").toBeTruthy(); + }); + + it("toggling pre-warm PUTs ultraSlmPrewarm:true when SLM is active", async () => { + const { puts } = setupFetchMock({ ultraEngine: "slm", ultraSlmPrewarm: false }); + const { default: CompressionPanel } = await import( + "../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel" + ); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const toggle = container.querySelector( + `[data-testid="ultra-slm-prewarm-toggle"] button, [data-testid="ultra-slm-prewarm-toggle"] input` + ) as HTMLElement | null; + expect(toggle, "pre-warm toggle must exist when slm preselected").toBeTruthy(); + await act(async () => { + toggle!.click(); + }); + await flush(); + + const put = puts.find((p) => "ultraSlmPrewarm" in p.body); + expect(put, "a PUT carrying ultraSlmPrewarm").toBeTruthy(); + expect(put!.body.ultraSlmPrewarm).toBe(true); + }); +}); From 8d10e042a9f4c8ddea97242c5201b1c7e675c2d5 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 19:39:23 -0300 Subject: [PATCH 16/18] test(compression): gated VPS ultra-SLM live + forced-fallback evidence (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- .../compression/llmlingua-ultra-entry.test.ts | 68 +++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/tests/unit/compression/llmlingua-ultra-entry.test.ts b/tests/unit/compression/llmlingua-ultra-entry.test.ts index cea4e1505b5..308aafd6cfe 100644 --- a/tests/unit/compression/llmlingua-ultra-entry.test.ts +++ b/tests/unit/compression/llmlingua-ultra-entry.test.ts @@ -104,3 +104,71 @@ test("prewarmLlmlinguaUltra is a no-op when unavailable", async () => { __resetUltraEntryForTests(); } }); + +// ─── Task 7 — gated VPS live validation (Hard Rule #18) ────────────────────── +// These tests run the REAL ONNX model and are SKIPPED unless RUN_LLMLINGUA_INT=1 +// AND the optional deps are present (only true on the VPS). Under the normal +// runner they print a skip line and pass as no-ops. +// +// VPS command (run ON the VPS, optional deps present, real ONNX model downloaded +// on the first call): +// +// RUN_LLMLINGUA_INT=1 node --import tsx --import ./open-sse/utils/setupPolyfill.ts \ +// --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit \ +// tests/unit/compression/llmlingua-ultra-entry.test.ts +// +// Expected: "GATED real ultra-SLM compression" PASSES with a real shrink + +// ultraTier:"slm"; "GATED forced-unavailable ultra falls back to heuristic" +// PASSES with ultraTier:"heuristic". +import { ultraCompress } from "../../../open-sse/services/compression/ultra.ts"; + +test("GATED real ultra-SLM compression (RUN_LLMLINGUA_INT=1)", async () => { + if (process.env.RUN_LLMLINGUA_INT !== "1") { + console.log("skip: RUN_LLMLINGUA_INT!=1"); + return; + } + if (!depsResolve()) { + console.log("skip: deps absent"); + return; + } + const LONG_PROSE = + "The quick brown fox jumps over the lazy dog while the sun sets slowly behind the distant hills. ".repeat( + 120 + ); + const r = await ultraCompress([{ role: "user", content: LONG_PROSE }], { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + ultraEngine: "slm", + }); + const out = r.messages[0].content as string; + assert.equal(typeof out, "string"); + assert.ok(out.length < LONG_PROSE.length, "expected a real SLM shrink"); + assert.equal(r.stats.ultraTier, "slm"); +}); + +test("GATED forced-unavailable ultra falls back to heuristic", async () => { + if (process.env.RUN_LLMLINGUA_INT !== "1") { + console.log("skip: RUN_LLMLINGUA_INT!=1"); + return; + } + __setUltraSlmTestHooks({ available: false }); + try { + const r = await ultraCompress( + [{ role: "user", content: "the quick brown fox jumps over the lazy dog ".repeat(40) }], + { + enabled: true, + compressionRate: 0.5, + minScoreThreshold: 0.3, + slmFallbackToAggressive: false, + maxTokensPerMessage: 0, + ultraEngine: "slm", + } + ); + assert.equal(r.stats.ultraTier, "heuristic"); + } finally { + __resetUltraEntryForTests(); + } +}); From 75d9f2c350dedc555a97210849d1f86930b683bc Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Mon, 22 Jun 2026 19:48:54 -0300 Subject: [PATCH 17/18] chore(compression): rebaseline file-size for ultra SLM tier strategySelector wiring (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- config/quality/file-size-baseline.json | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 46108fa8932..05a0066434e 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -106,7 +106,7 @@ "_rebaseline_2026_06_20_4389_thinking_toolchoice": "Re-baseline base.ts 1387->1399 (#4389): tool_choice-forced thinking guard at the existing Claude wire-image injection chokepoint (effThinking gate avoids the Anthropic 400 when tool_choice forces a tool). Cohesive guard; structural shrink tracked in #3501.", "cap": 800, "frozen": { - "open-sse/translator/request/openai-to-kiro.ts": 807, + "_rebaseline_pr1043_minimax_tts": "Upstream port decolua/9router#1043 (toanalien) own growth: audioSpeech.ts 965->1061 (+96). Adds MiniMax T2A v2 TTS dispatch (handleMinimaxSpeech + hexToBytes helper) — provider entry was already in audioRegistry (format: minimax-tts) but no handler existed, falling through to the OpenAI-compatible default that fails (T2A has custom shape + hex-encoded audio + base_resp envelope). New branch sits next to the other inline provider branches (xiaomi-mimo, coqui, tortoise, aws-polly) — extracting would just create indirection. Covered by tests/unit/minimax-tts-1043.test.ts (3 tests, GREEN: success, base_resp error, invalid-hex).", "open-sse/config/providerRegistry.ts": 4731, "open-sse/executors/antigravity.ts": 1696, "open-sse/executors/base.ts": 1414, @@ -120,7 +120,6 @@ "open-sse/executors/grok-web.ts": 1871, "open-sse/executors/muse-spark-web.ts": 1284, "open-sse/executors/perplexity-web.ts": 1013, - "_rebaseline_pr1043_minimax_tts": "Upstream port decolua/9router#1043 (toanalien) own growth: audioSpeech.ts 965->1061 (+96). Adds MiniMax T2A v2 TTS dispatch (handleMinimaxSpeech + hexToBytes helper) — provider entry was already in audioRegistry (format: minimax-tts) but no handler existed, falling through to the OpenAI-compatible default that fails (T2A has custom shape + hex-encoded audio + base_resp envelope). New branch sits next to the other inline provider branches (xiaomi-mimo, coqui, tortoise, aws-polly) — extracting would just create indirection. Covered by tests/unit/minimax-tts-1043.test.ts (3 tests, GREEN: success, base_resp error, invalid-hex).", "open-sse/handlers/audioSpeech.ts": 1061, "open-sse/handlers/chatCore.ts": 5125, "open-sse/handlers/imageGeneration.ts": 3777, @@ -137,10 +136,12 @@ "open-sse/services/claudeCodeCompatible.ts": 1202, "_rebaseline_pr4592_exclude_exhausted_auto": "Reconcile #4592 already-merged growth: combo.ts 2991->3036 (+45, terminal-status quota-cutoff exclusion in buildAutoCandidates + opt-in gate). Fast-gate PR->release does not run check:file-size.", "open-sse/services/combo.ts": 3036, + "open-sse/services/compression/strategySelector.ts": 818, "open-sse/services/rateLimitManager.ts": 1035, "open-sse/services/tokenRefresh.ts": 1997, "open-sse/services/usage.ts": 3450, "open-sse/translator/request/openai-to-gemini.ts": 864, + "open-sse/translator/request/openai-to-kiro.ts": 807, "open-sse/translator/response/openai-responses.ts": 922, "open-sse/utils/cursorAgentProtobuf.ts": 1521, "open-sse/utils/stream.ts": 2710, @@ -302,5 +303,6 @@ "_rebaseline_2026_06_20_4355_gpt5x_pro_pricing": "PR #4355 own growth: pricing.ts 1581->1592 (+11 = pure-data pricing rows for openai gpt-5.5-pro + gpt-5.4-pro, closing the $0 gap that tripped the catalog pricing gate after the #4324 sweep added them to the registry; -pro mirrors its base family tier). provider-models-route.test.ts 1616->1618 (+2 = test-only alignment to the intentional opencode-go discovery behavior: owned_by stamp + T39 two-endpoint fail-path fetchCalls). Both are data/test-only; not extractable.", "_rebaseline_2026_06_19_4293_codex_spark_scope": "PR #4293 (isolate Codex Spark quota scope) own growth, MEASURED on the actual merged tree (release/v3.8.30 + #4293). Production: auth.ts 2219->2279 (+60) threads requestedModel into Codex quota-policy/headroom/preflight/P2C scoring so normal Codex and GPT-5.3-Codex-Spark windows are evaluated independently; chatCore.ts 5116->5125 (+9) passes the failing model scope into Codex 429 failover (markCodexScopeRateLimited) instead of a connection-wide rateLimitedUntil write; accountFallback.ts 1727->1731 (+4) scopes Codex model-lock keys to codex vs spark. Heavy parsing/display logic lives in new leaf helpers under the cap (codexQuotaScopes.ts, codexUsageQuotas.ts, codexFailover.ts). Tests: account-fallback-service 1544->1569, executor-codex 1336->1339, sse-auth 1527->1553, usage-service-hardening 1612->1633 (added Spark-scope regression coverage). Cohesive wiring at existing selection/failover lockout boundaries; not extractable.", "_rebaseline_2026_06_20_4447_openai_gpt41mini_o_mini_pricing": "PR #4447 own growth: pricing.ts 1592->1620 (+28 = pure-data pricing rows closing the null/$0 gap for registry-exposed OpenAI ids gpt-4.1-mini, gpt-4.1-nano, o3-mini, o4-mini that tripped the catalog pricing gate; getPricingForModel does an exact lookup, so a missing key resolves to null. Official OpenAI per-1M prices + the table's derived-field convention (reasoning=output*1.5, cache_creation=input, cached=official). Restore-green for a pre-existing release/v3.8.32 red surfaced by #4432's __RUN_ALL__ run. Cohesive data; not extractable.", - "_rebaseline_2026_06_20_web_cookie_validator_shadow_fix": "validation.ts 4518->4522 (+4 = move the generic web-cookie validateWebCookieProvider dispatch from the TOP of validateProviderApiKey to a FALLBACK after SPECIALTY_VALIDATORS, plus a comment, so #4023's generic AUTH_007 ping no longer shadows the rich per-provider validators (grok-web #3474 IP-reputation/Cloudflare, chatgpt-web cf-mitigated, claude/gemini/copilot/qwen/t3-web). Restores provider-validation-specialty.test.ts (112/112) while keeping web-cookie-auth007 (5/5). Behavior fix at an existing dispatch boundary; not extractable." + "_rebaseline_2026_06_20_web_cookie_validator_shadow_fix": "validation.ts 4518->4522 (+4 = move the generic web-cookie validateWebCookieProvider dispatch from the TOP of validateProviderApiKey to a FALLBACK after SPECIALTY_VALIDATORS, plus a comment, so #4023's generic AUTH_007 ping no longer shadows the rich per-provider validators (grok-web #3474 IP-reputation/Cloudflare, chatgpt-web cf-mitigated, claude/gemini/copilot/qwen/t3-web). Restores provider-validation-specialty.test.ts (112/112) while keeping web-cookie-auth007 (5/5). Behavior fix at an existing dispatch boundary; not extractable.", + "_rebaseline_2026_06_22_phase4b_slm_tier_ultra": "Compression Phase 4 (B) SLM tier own growth: open-sse/services/compression/strategySelector.ts 783->818 (+35 at the existing applyUltraAsync chokepoint). The no-modelPath ultra branch (previously a one-line passthrough to the sync applyCompression) now runs the two-tier resolver: it adapts the body, builds the ultraConfig (threading config.ultraEngine + preserveSystemPrompt), awaits the now-async ultraCompress (SLM Tier-B when ultraEngine===slm and the worker backend is available, else fail-open to the Tier-A heuristic), and threads result.stats.ultraTier into the returned CompressionStats so the resolved tier reaches the D0 telemetry persister. The sync applyCompression ultra branch is also re-pointed to the new pure ultraCompressHeuristic. The two-tier resolver + the pure heuristic live in open-sse/services/compression/ultra.ts and the thin SLM entry in engines/llmlingua/ultraEntry.ts (both Date: Tue, 23 Jun 2026 07:08:40 -0300 Subject: [PATCH 18/18] test(compression): add useLocale to next-intl mock after A panel locale gate (B) Co-Authored-By: Claude Opus 4.8 (1M context) --- tests/unit/ui/compressionUltraTier.test.tsx | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/unit/ui/compressionUltraTier.test.tsx b/tests/unit/ui/compressionUltraTier.test.tsx index 00355a80c92..67cadedabea 100644 --- a/tests/unit/ui/compressionUltraTier.test.tsx +++ b/tests/unit/ui/compressionUltraTier.test.tsx @@ -6,7 +6,10 @@ import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; // i18n does not resolve to a real locale in vitest/jsdom, so mock next-intl to echo // the key. This test asserts ONLY on i18n-independent hooks (data-testid + values) // and the captured PUT body. -vi.mock("next-intl", () => ({ useTranslations: () => (key: string) => key })); +vi.mock("next-intl", () => ({ + useTranslations: () => (key: string) => key, + useLocale: () => "en", +})); const containers: HTMLElement[] = []; const roots: Array<{ unmount: () => void }> = [];