Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions config/quality/dependency-allowlist.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
{
"_comment": "Allowlist anti-slopsquatting (check-deps.mjs). Toda dep nova exige adicao EXPLICITA aqui apos verificar que e legitima.",
"allowed": [
"@atjsh/llmlingua-2",
"@aws-sdk/client-bedrock-runtime",
"@cyclonedx/cyclonedx-npm",
"@dnd-kit/core",
Expand All @@ -18,6 +19,7 @@
"@stryker-mutator/tap-runner",
"@swc/helpers",
"@tailwindcss/postcss",
"@tensorflow/tfjs",
"@testing-library/jest-dom",
"@testing-library/react",
"@types/bcryptjs",
Expand Down Expand Up @@ -64,6 +66,7 @@
"ink-text-input",
"ioredis",
"jose",
"js-tiktoken",
"js-yaml",
"jscpd",
"jsdom",
Expand Down
55 changes: 55 additions & 0 deletions open-sse/services/compression/engines/llmlingua/constants.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
/**
* LLMLingua real-engine constants — pure data + types.
*
* NO imports of native deps (transformers.js, onnxruntime, etc). This module is
* safe to import from anywhere (main thread, worker, tests) without pulling in
* the heavy ONNX runtime.
*
* The real backend uses `@atjsh/llmlingua-2` (ONNX via `@huggingface/transformers`),
* which downloads models from the HuggingFace Hub into a cache dir. Only the two
* models PROVEN to work end-to-end are registered here.
*/

export type LlmlinguaFactory = "WithBERTMultilingual" | "WithXLMRoBERTa";

export interface LlmlinguaModelEntry {
/** config value, e.g. "tinybert" */
id: string;
/** HuggingFace Hub repo id */
hfRepo: string;
factory: LlmlinguaFactory;
dtype: "fp32";
/** transformers.js subfolder option; "" for both proven models */
subfolder: string;
sizeMB: number;
label: string;
}

export const DEFAULT_LLMLINGUA_MODEL = "tinybert";

/** Registry keyed by config `model` value. Only the two PROVEN models. */
export const LLMLINGUA_MODELS: Record<string, LlmlinguaModelEntry> = {
tinybert: {
id: "tinybert",
hfRepo: "atjsh/llmlingua-2-js-tinybert-meetingbank",
factory: "WithBERTMultilingual",
dtype: "fp32",
subfolder: "",
sizeMB: 57,
label: "TinyBERT (57MB, fast — default)",
},
"bert-base": {
id: "bert-base",
hfRepo: "Arcoldd/llmlingua4j-bert-base-onnx",
factory: "WithBERTMultilingual",
dtype: "fp32",
subfolder: "",
sizeMB: 710,
label: "BERT-base (710MB, higher accuracy)",
},
};

/** Per-call worker reply timeout → fail-open. */
export const LLMLINGUA_WORKER_TIMEOUT_MS = 5000;
/** Terminate the idle worker after this long to free model RAM. */
export const LLMLINGUA_WORKER_IDLE_MS = 300000;
130 changes: 111 additions & 19 deletions open-sse/services/compression/engines/llmlingua/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,9 @@
* ## Design
*
* ### Backend abstraction
* `LlmlinguaBackend` is a simple `(text: string) => Promise<string>` contract.
* `LlmlinguaBackend` is a `(text: string, opts?: LlmlinguaBackendOptions) =>
* Promise<string>` contract (the opts carry model selection / compression rate /
* offline model-path override; single-arg fakes remain assignable).
* Tests inject a fake backend via `setLlmlinguaBackend()`. Production code uses
* `workerBackend` from `./worker.ts` (a stub today — see that file for the L1
* VPS-validation follow-up before the real ONNX model is wired).
Expand Down Expand Up @@ -42,7 +44,7 @@
* for the exact spec.
*/

import { createCompressionStats } from "../../stats.ts";
import { createCompressionStats, estimateCompressionTokens } from "../../stats.ts";
import { extractPreservedBlocks } from "../../preservation.ts";
import type {
CompressionEngine,
Expand All @@ -52,14 +54,22 @@ import type {
} from "../types.ts";
import type { CompressionResult } from "../../types.ts";
import { workerBackend } from "./worker.ts";
import { LLMLINGUA_MODELS, DEFAULT_LLMLINGUA_MODEL } from "./constants.ts";

// ─── backend abstraction ──────────────────────────────────────────────────────

/** Options the real backend needs (model selection + compression rate + offline override). */
export interface LlmlinguaBackendOptions {
model?: string;
compressionRate?: number;
modelPath?: string;
}

/**
* A backend takes a prose text segment and returns a compressed version.
* A backend takes a prose text segment (+ optional config) and returns a compressed version.
* Any rejection or error MUST be caught by the caller; the engine fail-opens.
*/
export type LlmlinguaBackend = (text: string) => Promise<string>;
export type LlmlinguaBackend = (text: string, opts?: LlmlinguaBackendOptions) => Promise<string>;

/** Module-level injectable backend (null = use default production backend). */
let _backend: LlmlinguaBackend | null = null;
Expand Down Expand Up @@ -140,11 +150,12 @@ type MessageLike = {
*/
async function compressProseText(
text: string,
backend: LlmlinguaBackend
backend: LlmlinguaBackend,
opts?: LlmlinguaBackendOptions
): Promise<{ text: string; didCompress: boolean }> {
if (!text.trim()) return { text, didCompress: false };
try {
const compressed = await backend(text);
const compressed = await backend(text, opts);
// Accept only if it actually gets shorter (reject no-ops or expansions)
if (typeof compressed === "string" && compressed.length < text.length) {
return { text: compressed, didCompress: true };
Expand All @@ -165,7 +176,8 @@ async function compressProseText(
*/
async function compressMessageText(
text: string,
backend: LlmlinguaBackend
backend: LlmlinguaBackend,
opts?: LlmlinguaBackendOptions
): Promise<{ text: string; didCompress: boolean }> {
const segments = splitProseAndPreserved(text);
let anyCompressed = false;
Expand All @@ -176,7 +188,7 @@ async function compressMessageText(
// Never send preserved content (code, math, etc.) to the backend
parts.push(seg.text);
} else {
const { text: out, didCompress } = await compressProseText(seg.text, backend);
const { text: out, didCompress } = await compressProseText(seg.text, backend, opts);
parts.push(out);
if (didCompress) anyCompressed = true;
}
Expand All @@ -191,7 +203,8 @@ async function compressMessageText(
*/
async function processMessages(
messages: MessageLike[],
backend: LlmlinguaBackend
backend: LlmlinguaBackend,
opts?: LlmlinguaBackendOptions
): Promise<{ messages: MessageLike[]; compressedCount: number }> {
let compressedCount = 0;
const result: MessageLike[] = [];
Expand All @@ -205,7 +218,7 @@ async function processMessages(

try {
if (typeof msg.content === "string") {
const { text, didCompress } = await compressMessageText(msg.content, backend);
const { text, didCompress } = await compressMessageText(msg.content, backend, opts);
if (didCompress) {
compressedCount++;
result.push({ ...msg, content: text });
Expand All @@ -219,7 +232,8 @@ async function processMessages(
if (part["type"] === "text" && typeof part["text"] === "string") {
const { text, didCompress } = await compressMessageText(
part["text"] as string,
backend
backend,
opts
);
if (didCompress) {
changed = true;
Expand Down Expand Up @@ -248,19 +262,65 @@ async function processMessages(
// ─── config schema ────────────────────────────────────────────────────────────

const LLMLINGUA_SCHEMA: EngineConfigField[] = [
{ key: "enabled", type: "boolean", label: "Enabled", defaultValue: true },
{
key: "model",
type: "select",
label: "Model",
defaultValue: DEFAULT_LLMLINGUA_MODEL,
options: Object.values(LLMLINGUA_MODELS).map((m) => ({ value: m.id, label: m.label })),
},
{
key: "minTokens",
type: "number",
label: "Min tokens (floor)",
defaultValue: 2000,
min: 0,
max: 100000,
},
{
key: "enabled",
type: "boolean",
label: "Enabled",
defaultValue: true,
key: "compressionRate",
type: "number",
label: "Compression rate (keep ratio)",
defaultValue: 0.5,
min: 0.1,
max: 0.9,
},
{ key: "modelPath", type: "string", label: "Model path (offline override)", defaultValue: "" },
];

function validateLlmlinguaConfig(config: Record<string, unknown>): EngineValidationResult {
const errors: string[] = [];

if (config["enabled"] !== undefined && typeof config["enabled"] !== "boolean") {
errors.push("enabled must be a boolean");
}

if (config["model"] !== undefined) {
const model = config["model"];
if (typeof model !== "string" || !(model in LLMLINGUA_MODELS)) {
errors.push("model must be one of: " + Object.keys(LLMLINGUA_MODELS).join(", "));
}
}

if (config["minTokens"] !== undefined) {
const minTokens = config["minTokens"];
if (typeof minTokens !== "number" || Number.isNaN(minTokens) || minTokens < 0) {
errors.push("minTokens must be a number >= 0");
}
}

if (config["compressionRate"] !== undefined) {
const rate = config["compressionRate"];
if (typeof rate !== "number" || Number.isNaN(rate) || rate < 0.1 || rate > 0.9) {
errors.push("compressionRate must be a number between 0.1 and 0.9");
}
}

if (config["modelPath"] !== undefined && typeof config["modelPath"] !== "string") {
errors.push("modelPath must be a string");
}

return { valid: errors.length === 0, errors };
}

Expand All @@ -275,8 +335,8 @@ export const llmlinguaEngine: CompressionEngine = {
"Async semantic token pruning via LLMLingua-2 (ONNX/worker-thread backend). " +
"Compresses prose in non-system messages; fenced code blocks and other preserved " +
"constructs are never altered. Fail-opens on any backend error. Production backend: " +
"vendored @atjsh/llmlingua-2 (MobileBERT 99 MB) in a worker thread — see " +
"./worker.ts for the L1 follow-up spec (VPS validation required per Hard Rule #18).",
"@atjsh/llmlingua-2 (TinyBERT 57 MB default, BERT-base optional) in a worker thread; " +
"model lazy-downloaded to DATA_DIR. Optional deps — fail-opens if not installed.",
icon: "brain",
targets: ["messages"],
stackable: true,
Expand All @@ -294,7 +354,10 @@ export const llmlinguaEngine: CompressionEngine = {
inputScope: "messages",
targetLatencyMs: 200,
supportsPreview: false,
stable: false,
// Promoted to stable after VPS validation (2026-06-16): the deployed worker
// compressed real prose (209→107 ch, ok=true), and the bundle's walk-up
// resolution + optional-deps gate were confirmed against the live install.
stable: true,
},

/**
Expand Down Expand Up @@ -334,12 +397,41 @@ export const llmlinguaEngine: CompressionEngine = {
return { body, compressed: false, stats: null };
}

// minTokens floor: skip the model entirely on small prompts (avoid paying
// model latency when there is little to gain). 0 disables the floor.
const minTokens =
typeof stepConfig["minTokens"] === "number" ? (stepConfig["minTokens"] as number) : 2000;
if (minTokens > 0) {
const nonSystemText = (messages as MessageLike[])
.filter((m) => m.role !== "system")
.map((m) => (typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? "")))
.join("\n");
Comment on lines +405 to +408

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

When m.content is an array of message parts (e.g., in multi-part messages containing text and other media), using JSON.stringify on the entire array includes JSON syntax boilerplate (brackets, keys, quotes, etc.). This artificially inflates the estimated token count calculated by estimateCompressionTokens, which can cause the engine to incorrectly skip or apply the token floor. Extracting only the actual text parts from the array (similar to how processMessages handles it) provides a much more accurate token estimation.

      const nonSystemText = (messages as MessageLike[])
        .filter((m) => m.role !== "system")
        .map((m) => {
          if (typeof m.content === "string") return m.content;
          if (Array.isArray(m.content)) {
            return m.content
              .map((part) => (part && typeof part === "object" && part.type === "text" && typeof part.text === "string" ? part.text : ""))
              .join("\n");
          }
          return "";
        })
        .join("\n");

if (estimateCompressionTokens(nonSystemText) < minTokens) {
// Below the floor — skip compression for this small prompt.
return { body, compressed: false, stats: null };
}
}

// Backend options threaded from stepConfig (model selection / rate / offline override).
const backendOpts: LlmlinguaBackendOptions = {
model: typeof stepConfig["model"] === "string" ? (stepConfig["model"] as string) : undefined,
compressionRate:
typeof stepConfig["compressionRate"] === "number"
? (stepConfig["compressionRate"] as number)
: undefined,
modelPath:
typeof stepConfig["modelPath"] === "string" && stepConfig["modelPath"]
? (stepConfig["modelPath"] as string)
: undefined,
};

try {
const backend = resolveBackend();
const start = performance.now();
const { messages: newMessages, compressedCount } = await processMessages(
messages as MessageLike[],
backend
backend,
backendOpts
);

if (compressedCount === 0) {
Expand Down
67 changes: 67 additions & 0 deletions open-sse/services/compression/engines/llmlingua/modelStore.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
/**
* LLMLingua model store — thin path/config resolver.
*
* transformers.js owns the actual model download (from the HuggingFace Hub into
* its `cacheDir`). This module only resolves the cache directory, maps config
* model ids to registry entries, and configures a transformers.js `env` object
* for either Hub download (default) or a local modelPath override.
*
* Deliberately does NOT import the native `@huggingface/transformers` dep — it
* accepts a minimal structural `env` so the heavy runtime stays out of this path.
*/

import os from "node:os";
import path from "node:path";
import fs from "node:fs";
import {
DEFAULT_LLMLINGUA_MODEL,
LLMLINGUA_MODELS,
type LlmlinguaModelEntry,
} from "./constants.ts";

/** A minimal structural type for the transformers.js `env` object (avoids importing the native dep here). */
export interface TransformersEnvLike {
cacheDir?: string;
localModelPath?: string;
allowRemoteModels?: boolean;
[key: string]: unknown;
}

/** Base data dir. Mirrors rtk's getDataDir() at engines/rtk/filterLoader.ts. */
function getDataDir(): string {
return process.env.DATA_DIR || path.join(os.homedir(), ".omniroute");
}

/** Resolve (and ensure) the model cache dir: `${DATA_DIR}/models/llmlingua`. Mirrors rtk's getDataDir(). */
export function getLlmlinguaModelCacheDir(): string {
const dir = path.join(getDataDir(), "models", "llmlingua");
try {
fs.mkdirSync(dir, { recursive: true });
} catch {
// Ignore mkdir errors — fail-open philosophy: transformers.js will surface a
// clearer error if the dir is genuinely unusable, and callers fail-open anyway.
}
return dir;
}

/** Resolve a config model id to its registry entry; falls back to the default for unknown/empty ids. */
export function resolveLlmlinguaModel(modelId: string | undefined | null): LlmlinguaModelEntry {
if (typeof modelId === "string" && modelId.length > 0 && LLMLINGUA_MODELS[modelId]) {
return LLMLINGUA_MODELS[modelId];
}
return LLMLINGUA_MODELS[DEFAULT_LLMLINGUA_MODEL];
}

/** Configure a transformers.js `env` for either Hub download (default) or a local modelPath override. */
export function configureTransformersEnv(
env: TransformersEnvLike,
opts: { modelPath?: string }
): void {
env.cacheDir = getLlmlinguaModelCacheDir();
if (typeof opts.modelPath === "string" && opts.modelPath.length > 0) {
env.localModelPath = opts.modelPath;
env.allowRemoteModels = false;
} else {
env.allowRemoteModels = true;
}
}
Loading