diff --git a/open-sse/handlers/ollamaSystemOne.ts b/open-sse/handlers/ollamaSystemOne.ts new file mode 100644 index 00000000000..ae569372a73 --- /dev/null +++ b/open-sse/handlers/ollamaSystemOne.ts @@ -0,0 +1,412 @@ +/** + * Ollama System One proxy (`POST {ollama}/v1/systemone`, Ollama >= 0.35). + * + * System One models (Nimble, Tev) answer typed `choice` / `noul` / `score` + * questions about a `state` with calibrated probabilities instead of generating + * text. The request is a single non-streaming JSON POST, so this handler only + * validates, forwards the known fields, and maps failures onto OmniRoute's + * resilience layers: + * + * - 404 (model not pulled) → model lockout on that connection + * - 400 (e.g. not a System One model) → returned as-is, connection untouched + * - 5xx / unreachable / timeout → connection cooldown + * + * Limits mirror the ones Ollama enforces so callers get the same answers + * without a round trip. See https://docs.ollama.com/api/systemone + */ + +import { z } from "zod"; +import { CORS_HEADERS } from "../utils/cors.ts"; +import { errorResponse } from "../utils/error.ts"; +import { stripTrailingSlashes } from "../utils/urlSanitize.ts"; +import { attachOmniRouteMetaHeaders } from "@/domain/omnirouteResponseMeta"; +import { generateRequestId } from "@/shared/utils/requestId"; +import { saveCallLog } from "@/lib/usageDb"; + +export const OLLAMA_SYSTEMONE_PROVIDER = "ollama-local"; +export const OLLAMA_SYSTEMONE_DEFAULT_BASE_URL = "http://localhost:11434/v1"; +export const OLLAMA_SYSTEMONE_MAX_BODY_BYTES = 64 * 1024; +// Bodies carrying `images` (Clef / Clef Flash vision weights) may be up to 32 MiB. +export const OLLAMA_SYSTEMONE_MAX_IMAGE_BODY_BYTES = 32 * 1024 * 1024; +export const OLLAMA_SYSTEMONE_MAX_QUESTIONS = 64; +export const OLLAMA_SYSTEMONE_MIN_CRITERIA = 2; +export const OLLAMA_SYSTEMONE_MAX_CRITERIA = 26; +// The first call loads the model; on a Jetson-class host that alone took 8–10s. +export const OLLAMA_SYSTEMONE_DEFAULT_TIMEOUT_MS = 60_000; + +const UNSUPPORTED_MODEL_PATTERN = /not supported by system one/i; + +const structuredTextSchema = z.union([ + z.string().trim().min(1), + z.record(z.string(), z.unknown()), + z.array(z.unknown()), +]); + +const criteriaCountMessage = `criteria must contain ${OLLAMA_SYSTEMONE_MIN_CRITERIA}–${OLLAMA_SYSTEMONE_MAX_CRITERIA} candidates`; + +const choiceQuestionSchema = z + .object({ + type: z.literal("choice"), + instructions: structuredTextSchema, + criteria: z.record(z.string().min(1), z.unknown()).refine((criteria) => { + const count = Object.keys(criteria).length; + return count >= OLLAMA_SYSTEMONE_MIN_CRITERIA && count <= OLLAMA_SYSTEMONE_MAX_CRITERIA; + }, criteriaCountMessage), + }) + .passthrough(); + +const noulQuestionSchema = z + .object({ + type: z.literal("noul"), + instructions: structuredTextSchema, + criteria: z + .object({ true: z.unknown().optional(), false: z.unknown().optional() }) + .passthrough() + .optional(), + }) + .passthrough(); + +const scoreQuestionSchema = z + .object({ + type: z.literal("score"), + instructions: structuredTextSchema, + criteria: z + .array(z.unknown()) + .min(OLLAMA_SYSTEMONE_MIN_CRITERIA, criteriaCountMessage) + .max(OLLAMA_SYSTEMONE_MAX_CRITERIA, criteriaCountMessage), + }) + .passthrough(); + +// Ollama takes raw base64 only: "URLs and data URLs are not supported". +const imageSchema = z + .string() + .min(1, "images must be non-empty base64 strings") + .refine( + (value) => !/^(?:https?:|data:)/i.test(value.trim()), + "images must be raw base64; URLs and data URLs are not supported" + ); + +const questionSchema = z.discriminatedUnion("type", [ + choiceQuestionSchema, + noulQuestionSchema, + scoreQuestionSchema, +]); + +export const ollamaSystemOneRequestSchema = z.object({ + model: z.string().trim().min(1), + state: structuredTextSchema, + questions: z.record(z.string().min(1), questionSchema).refine((questions) => { + const count = Object.keys(questions).length; + return count >= 1 && count <= OLLAMA_SYSTEMONE_MAX_QUESTIONS; + }, `questions must contain 1–${OLLAMA_SYSTEMONE_MAX_QUESTIONS} fields`), + images: z.array(imageSchema).optional(), + keep_alive: z.union([z.string().min(1), z.number()]).optional(), +}); + +export type OllamaSystemOneRequest = z.infer; + +export type OllamaSystemOneValidation = + { ok: true; data: OllamaSystemOneRequest } | { ok: false; status: 400 | 413; message: string }; + +/** + * Validate a raw `/v1/systemone` body against Ollama's limits. Unknown top-level + * fields (e.g. `stream`) are dropped, matching Ollama, which ignores them. + */ +export function validateOllamaSystemOneRequest(raw: unknown): OllamaSystemOneValidation { + let serializedBytes = 0; + try { + serializedBytes = Buffer.byteLength(JSON.stringify(raw ?? null), "utf8"); + } catch { + return { ok: false, status: 400, message: "Invalid JSON body" }; + } + const images = (raw as { images?: unknown } | null)?.images; + const hasImages = Array.isArray(images) && images.length > 0; + if (hasImages && serializedBytes > OLLAMA_SYSTEMONE_MAX_IMAGE_BODY_BYTES) { + return { ok: false, status: 413, message: "request body with images must not exceed 32 MiB" }; + } + if (!hasImages && serializedBytes > OLLAMA_SYSTEMONE_MAX_BODY_BYTES) { + return { ok: false, status: 413, message: "request body must not exceed 64 KiB" }; + } + + const parsed = ollamaSystemOneRequestSchema.safeParse(raw); + if (!parsed.success) { + const issue = parsed.error.issues[0]; + const where = issue?.path?.length ? `${issue.path.join(".")}: ` : ""; + return { ok: false, status: 400, message: `${where}${issue?.message ?? "invalid request"}` }; + } + return { ok: true, data: parsed.data }; +} + +/** Build the native System One URL from the connection's OpenAI-compatible base URL. */ +export function buildOllamaSystemOneUrl(baseUrl: string | null | undefined): string { + let base = stripTrailingSlashes((baseUrl || OLLAMA_SYSTEMONE_DEFAULT_BASE_URL).trim()); + base = base.replace(/\/(?:chat\/completions|completions|embeddings|systemone)$/i, ""); + base = base.replace(/\/api\/chat$/i, ""); + if (base.toLowerCase().endsWith("/v1")) base = base.slice(0, -3); + return `${base}/v1/systemone`; +} + +export type OllamaSystemOneFailureKind = + | "model_not_found" + | "unsupported_model" + | "invalid_request" + | "upstream_error" + | "unreachable" + | "timeout"; + +/** Map an upstream status onto the resilience layer it belongs to. */ +export function classifyOllamaSystemOneFailure( + status: number, + message: string +): OllamaSystemOneFailureKind { + if (status === 404) return "model_not_found"; + if (status === 400 && UNSUPPORTED_MODEL_PATTERN.test(message)) return "unsupported_model"; + if (status >= 500) return "upstream_error"; + return "invalid_request"; +} + +/** Only these kinds say something about the connection or model, not the request. */ +function shouldMarkUnavailable(kind: OllamaSystemOneFailureKind): boolean { + return ( + kind === "model_not_found" || + kind === "upstream_error" || + kind === "unreachable" || + kind === "timeout" + ); +} + +export interface OllamaSystemOneCredentials { + connectionId?: string | null; + apiKey?: string | null; + providerSpecificData?: { baseUrl?: unknown } | null; +} + +type MarkUnavailable = ( + connectionId: string, + status: number, + errorText: string, + provider: string | null, + model: string | null +) => Promise; + +export interface OllamaSystemOneOptions { + /** Validated body; `model` is the upstream id without the provider prefix. */ + body: OllamaSystemOneRequest; + /** Model id as the caller sent it (e.g. `ollama-local/nimble`), echoed back. */ + requestedModel: string; + credentials: OllamaSystemOneCredentials | null; + provider?: string; + timeoutMs?: number; + signal?: AbortSignal | null; + apiKeyId?: string | null; + apiKeyName?: string | null; + fetchImpl?: typeof fetch; + markAccountUnavailable?: MarkUnavailable; + clearRecoveredState?: (credentials: OllamaSystemOneCredentials) => Promise; + logCall?: (entry: Record) => Promise; +} + +async function defaultClearRecoveredState( + credentials: OllamaSystemOneCredentials +): Promise { + const { clearRecoveredProviderState } = await import("@/sse/services/auth.ts"); + return clearRecoveredProviderState(credentials); +} + +async function defaultMarkAccountUnavailable( + ...args: Parameters +): Promise { + const { markAccountUnavailable } = await import("@/sse/services/auth.ts"); + return markAccountUnavailable(...args); +} + +function readUpstreamError(parsed: unknown, text: string, status: number): string { + const record = parsed && typeof parsed === "object" ? (parsed as Record) : null; + const error = record?.error; + if (typeof error === "string" && error) return error; + if (error && typeof error === "object") { + const message = (error as Record).message; + if (typeof message === "string" && message) return message; + } + return text.slice(0, 500) || `Ollama returned HTTP ${status}`; +} + +function readUsage(parsed: Record): { input: number; output: number } { + const usage = + parsed.usage && typeof parsed.usage === "object" + ? (parsed.usage as Record) + : {}; + const asCount = (value: unknown) => + typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : 0; + return { input: asCount(usage.input_tokens), output: asCount(usage.output_tokens) }; +} + +function buildUpstreamBody(body: OllamaSystemOneRequest): Record { + return { + model: body.model, + state: body.state, + questions: body.questions, + ...(body.images?.length ? { images: body.images } : {}), + ...(body.keep_alive !== undefined ? { keep_alive: body.keep_alive } : {}), + }; +} + +function buildUpstreamHeaders(credentials: OllamaSystemOneCredentials | null) { + const headers: Record = { + "Content-Type": "application/json", + Accept: "application/json", + }; + // Ollama itself has no auth; a key is only meaningful behind an auth proxy. + if (credentials?.apiKey) headers.Authorization = `Bearer ${credentials.apiKey}`; + return headers; +} + +function resolveUpstreamUrl(credentials: OllamaSystemOneCredentials | null): string { + const configuredBaseUrl = credentials?.providerSpecificData?.baseUrl; + return buildOllamaSystemOneUrl(typeof configuredBaseUrl === "string" ? configuredBaseUrl : null); +} + +type FetchFailure = + { kind: "aborted" } | { kind: OllamaSystemOneFailureKind; status: number; message: string }; + +/** Why the fetch threw: the caller left, the timeout fired, or the host is unreachable. */ +function classifyFetchFailure( + clientSignal: AbortSignal | null | undefined, + timeoutSignal: AbortSignal, + timeoutMs: number +): FetchFailure { + // The caller going away says nothing about the connection. + if (clientSignal?.aborted) return { kind: "aborted" }; + if (timeoutSignal.aborted) { + const seconds = Math.round(timeoutMs / 1000); + return { + kind: "timeout", + status: 504, + message: `Ollama System One did not answer within ${seconds}s`, + }; + } + return { kind: "unreachable", status: 503, message: "Ollama System One upstream is unreachable" }; +} + +function parseJsonText(text: string): unknown { + try { + return text ? JSON.parse(text) : null; + } catch { + return null; + } +} + +/** The answers payload, or the reason a 200 is not a usable System One response. */ +function readAnswersPayload( + parsed: unknown +): { record: Record } | { error: string } { + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) { + return { error: "Ollama System One returned an invalid response" }; + } + const record = parsed as Record; + if (!record.answers || typeof record.answers !== "object") { + return { error: "Ollama System One response is missing answers" }; + } + return { record }; +} + +async function clearRecoveredConnection(options: OllamaSystemOneOptions): Promise { + if (!options.credentials?.connectionId) return; + try { + await (options.clearRecoveredState ?? defaultClearRecoveredState)(options.credentials); + } catch { + // Best effort, same as the other non-chat proxies. + } +} + +function buildSuccessResponse( + record: Record, + provider: string, + requestedModel: string, + startTime: number +): Response { + const responseHeaders = new Headers({ ...CORS_HEADERS, "Content-Type": "application/json" }); + attachOmniRouteMetaHeaders(responseHeaders, { + provider, + model: requestedModel, + costUsd: 0, + latencyMs: Date.now() - startTime, + requestId: generateRequestId(), + }); + // Ollama echoes `nimble` for `nimble:latest`; report the id the caller routed with. + return new Response(JSON.stringify({ ...record, model: requestedModel }), { + status: 200, + headers: responseHeaders, + }); +} + +export async function handleOllamaSystemOne(options: OllamaSystemOneOptions): Promise { + const startTime = Date.now(); + const provider = options.provider || OLLAMA_SYSTEMONE_PROVIDER; + const connectionId = options.credentials?.connectionId || null; + const markUnavailable = options.markAccountUnavailable ?? defaultMarkAccountUnavailable; + const logCall = options.logCall ?? saveCallLog; + + const log = (status: number, extra: Record = {}) => { + logCall({ + method: "POST", + path: "/v1/systemone", + status, + model: options.requestedModel, + provider, + duration: Date.now() - startTime, + connectionId, + apiKeyId: options.apiKeyId ?? null, + apiKeyName: options.apiKeyName ?? null, + ...extra, + }).catch(() => {}); + }; + + const fail = async (kind: OllamaSystemOneFailureKind, status: number, message: string) => { + log(status, { error: message.slice(0, 500) }); + if (connectionId && shouldMarkUnavailable(kind)) { + try { + await markUnavailable(connectionId, status, message, provider, options.body.model); + } catch { + // Resilience bookkeeping must never mask the upstream answer. + } + } + return errorResponse(status, message); + }; + + const timeoutMs = options.timeoutMs ?? OLLAMA_SYSTEMONE_DEFAULT_TIMEOUT_MS; + const timeoutSignal = AbortSignal.timeout(timeoutMs); + const signal = options.signal ? AbortSignal.any([options.signal, timeoutSignal]) : timeoutSignal; + + let res: Response; + try { + res = await (options.fetchImpl ?? fetch)(resolveUpstreamUrl(options.credentials), { + method: "POST", + headers: buildUpstreamHeaders(options.credentials), + body: JSON.stringify(buildUpstreamBody(options.body)), + signal, + }); + } catch { + const failure = classifyFetchFailure(options.signal, timeoutSignal, timeoutMs); + if (failure.kind === "aborted") { + log(499, { error: "client aborted" }); + return errorResponse(499, "Request aborted by client"); + } + return fail(failure.kind, failure.status, failure.message); + } + + const text = await res.text().catch(() => ""); + const parsed = parseJsonText(text); + if (!res.ok) { + const message = readUpstreamError(parsed, text, res.status); + return fail(classifyOllamaSystemOneFailure(res.status, message), res.status, message); + } + + const payload = readAnswersPayload(parsed); + if ("error" in payload) return fail("upstream_error", 502, payload.error); + + const usage = readUsage(payload.record); + log(200, { tokens: { prompt_tokens: usage.input, completion_tokens: usage.output } }); + await clearRecoveredConnection(options); + return buildSuccessResponse(payload.record, provider, options.requestedModel, startTime); +} diff --git a/open-sse/services/modelEndpointPolicy.ts b/open-sse/services/modelEndpointPolicy.ts index b46d079d978..0f4930f8728 100644 --- a/open-sse/services/modelEndpointPolicy.ts +++ b/open-sse/services/modelEndpointPolicy.ts @@ -8,7 +8,7 @@ */ export type ModelEndpointKind = - "chat" | "image" | "video" | "embedding" | "rerank" | "non-chat" | "unknown"; + "chat" | "image" | "video" | "embedding" | "rerank" | "decision" | "non-chat" | "unknown"; export type ModelEndpointDecision = { kind: ModelEndpointKind; @@ -32,6 +32,9 @@ const EMBEDDING_ENDPOINTS = new Set(["embeddings", "embedding"]); const RERANK_ENDPOINTS = new Set(["rerank", "reranking"]); const IMAGE_ENDPOINTS = new Set(["image", "images", "images/generations"]); const VIDEO_ENDPOINTS = new Set(["video", "videos", "videos/generations"]); +// System One decision models (Ollama Clef / Clef Flash, TypeSafe Jev): typed questions in, +// probabilities out — never a chat completion. +const DECISION_ENDPOINTS = new Set(["systemone"]); function normalizeEndpoint(endpoint: string): string { return endpoint.trim().toLowerCase().replace(/^\/+/, "").replace(/^v1\//, ""); @@ -58,6 +61,9 @@ function classifyExplicitEndpoints( if (endpoints.some((endpoint) => VIDEO_ENDPOINTS.has(endpoint))) { return { kind: "video", chatSelectable: false, reason: "explicit-endpoints" }; } + if (endpoints.some((endpoint) => DECISION_ENDPOINTS.has(endpoint))) { + return { kind: "decision", chatSelectable: false, reason: "explicit-endpoints" }; + } return { kind: "non-chat", chatSelectable: false, reason: "explicit-endpoints" }; } diff --git a/src/lib/providerModels/decisionOnlyChatGuard.ts b/src/lib/providerModels/decisionOnlyChatGuard.ts new file mode 100644 index 00000000000..b864e4ba4a0 --- /dev/null +++ b/src/lib/providerModels/decisionOnlyChatGuard.ts @@ -0,0 +1,42 @@ +import { getModelEndpointDecision } from "@omniroute/open-sse/services/modelEndpointPolicy.ts"; +import { errorResponse } from "@omniroute/open-sse/utils/error.ts"; +import { getSyncedAvailableModels, type SyncedAvailableModel } from "@/lib/db/models"; + +type DecisionOnlyChatGuardDeps = { + getSyncedAvailableModels: (providerId: string) => Promise; +}; + +/** + * Providers whose discovery can record a System One (`decision`) model. Only these pay the + * stored-model lookup on the chat path; every other provider is untouched. + */ +const SYSTEM_ONE_CAPABLE_PROVIDERS = new Set(["ollama-local"]); + +/** + * Refuse a chat call to a model that only serves `/v1/systemone`. + * + * Ollama's `/api/show` reports Clef / Clef Flash as `["vision", "decision"]` with no + * `completion`; discovery stores that as `supportedEndpoints: ["systemone"]`. Sending + * such a model a chat request only earns an upstream 400, so it is refused here, before + * any credential is selected — nothing is marked on the connection. The message reads as + * a model-scoped 400, so a combo advances to its next target instead of stopping. + */ +export async function decisionOnlyChatRejection( + modelInfo: { provider?: string | null; model?: string | null }, + deps: DecisionOnlyChatGuardDeps = { getSyncedAvailableModels } +): Promise { + const provider = modelInfo.provider ?? ""; + const model = modelInfo.model ?? ""; + if (!model || !SYSTEM_ONE_CAPABLE_PROVIDERS.has(provider)) return null; + + const row = (await deps.getSyncedAvailableModels(provider)).find((m) => m.id === model); + if (!row) return null; + if (getModelEndpointDecision(provider, model, row.supportedEndpoints).kind !== "decision") { + return null; + } + return errorResponse( + 400, + `Model ${provider}/${model} does not support chat: it is a System One decision model; ` + + "use the System One API (POST v1/systemone)" + ); +} diff --git a/src/lib/providerModels/ollamaCapabilities.ts b/src/lib/providerModels/ollamaCapabilities.ts index e3eaa73129f..17ebdbc0988 100644 --- a/src/lib/providerModels/ollamaCapabilities.ts +++ b/src/lib/providerModels/ollamaCapabilities.ts @@ -12,8 +12,17 @@ const OLLAMA_CAPABILITY_TO_ENDPOINT: Readonly> = { completion: "chat", embedding: "embeddings", image: "images", + // Ollama >= 0.35 tags System One models (Nimble, Tev) with `decision`; they answer + // typed questions on `/v1/systemone`. + decision: "systemone", }; +const ENDPOINT_TO_API_FORMAT: ReadonlyArray = [ + ["chat", "chat-completions"], + ["embeddings", "embeddings"], + ["images", "images-generations"], +]; + const MAX_CONCURRENT_SHOW_REQUESTS = 4; function asRecord(value: unknown): JsonRecord { @@ -45,15 +54,15 @@ export function applyOllamaShowCapabilities(model: unknown, showResponse: unknow ); if (supportedEndpoints.length === 0) return record; - const apiFormat = supportedEndpoints.includes("chat") - ? "chat-completions" - : supportedEndpoints.includes("embeddings") - ? "embeddings" - : "images-generations"; + // `systemone` has no wire format of its own, so a decision-only model keeps whatever + // apiFormat the record already carried. + const apiFormat = ENDPOINT_TO_API_FORMAT.find(([endpoint]) => + supportedEndpoints.includes(endpoint) + )?.[1]; return { ...record, - apiFormat, + ...(apiFormat ? { apiFormat } : {}), supportedEndpoints, ...(capabilities.includes("vision") ? { supportsVision: true } : {}), ...(capabilities.includes("tools") ? { supportsTools: true } : {}), diff --git a/src/shared/constants/modelSupportedEndpoints.ts b/src/shared/constants/modelSupportedEndpoints.ts index 14aa2b3b35d..9dcc31dc274 100644 --- a/src/shared/constants/modelSupportedEndpoints.ts +++ b/src/shared/constants/modelSupportedEndpoints.ts @@ -7,6 +7,8 @@ export const MODEL_SUPPORTED_ENDPOINT_VALUES = [ "audio-speech", "audio-transcriptions", "images-generations", + // System One typed-decision API (`POST /v1/systemone`): choice / noul / score answers. + "systemone", // Persisted legacy values remain valid input and normalize on write/edit. "video", "audio", @@ -33,14 +35,21 @@ export function normalizeModelSupportedEndpoints(endpoints: readonly string[]): return normalized; } +/** A System One model that does not also serve chat is a decision model (Clef, Clef Flash). */ +function isDecisionOnly(endpoints: readonly string[]): boolean { + if (!endpoints.includes("systemone")) return false; + return !endpoints.includes("chat") && !endpoints.includes("responses"); +} + export function classifyModelSupportedEndpoints(endpoints: readonly string[]): { - type?: "embedding" | "rerank" | "image" | "video" | "audio"; + type?: "embedding" | "rerank" | "image" | "video" | "audio" | "decision"; subtype?: "speech" | "transcription"; } { if (endpoints.includes("embeddings")) return { type: "embedding" }; if (endpoints.includes("rerank")) return { type: "rerank" }; if (endpoints.includes("images")) return { type: "image" }; if (endpoints.includes("videos") || endpoints.includes("video")) return { type: "video" }; + if (isDecisionOnly(endpoints)) return { type: "decision" }; const supportsSpeech = endpoints.includes("audio-speech"); const supportsTranscription = diff --git a/src/sse/services/model.ts b/src/sse/services/model.ts index 8a05aa6897a..ac889accdd4 100644 --- a/src/sse/services/model.ts +++ b/src/sse/services/model.ts @@ -33,6 +33,7 @@ import { import { commonChatGptWebRetirementResponse } from "@/lib/providers/chatgptWebRetirementResponse"; import { errorResponse } from "@omniroute/open-sse/utils/error.ts"; import { HTTP_STATUS } from "@omniroute/open-sse/config/constants.ts"; +import { decisionOnlyChatRejection } from "@/lib/providerModels/decisionOnlyChatGuard"; export { parseModel, stripContextWindowSuffix }; @@ -614,8 +615,9 @@ export async function getModelInfo(modelStr) { } export async function getModelInfoOrRetirementResponse(modelId: string) { + let modelInfo: Awaited>; try { - return await getModelInfo(modelId); + modelInfo = await getModelInfo(modelId); } catch (error) { if (isMicrosoftDesignerWebProviderRetiredError(error)) { return { error: errorResponse(HTTP_STATUS.GONE, error.message) }; @@ -633,6 +635,10 @@ export async function getModelInfoOrRetirementResponse(modelId: string) { } throw error; } + // A System One decision model (Clef / Clef Flash) cannot serve chat: refuse it here, + // before any credential is picked, with a 400 a combo treats as model-scoped. + const decisionOnly = await decisionOnlyChatRejection(modelInfo); + return decisionOnly ? { error: decisionOnly } : modelInfo; } /** diff --git a/stryker.conf.json b/stryker.conf.json index 721484f9d63..1016a5cc57d 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -570,7 +570,8 @@ "tests/unit/search-success-clears-connection-error.test.ts", "tests/unit/secret-compare-constant-time.test.ts", "tests/unit/version-manager-local-only.test.ts", - "tests/unit/stream-failure-after-output-no-model-lockout.test.ts" + "tests/unit/stream-failure-after-output-no-model-lockout.test.ts", + "tests/unit/ollama-systemone.test.ts" ], "nodeArgs": [ "--import", diff --git a/tests/unit/ollama-systemone-decision-guard.test.ts b/tests/unit/ollama-systemone-decision-guard.test.ts new file mode 100644 index 00000000000..e6a7edf654a --- /dev/null +++ b/tests/unit/ollama-systemone-decision-guard.test.ts @@ -0,0 +1,141 @@ +/** + * Decision-only models (Ollama System One: Clef, Clef Flash) must not be served as chat. + * + * Ollama's `/api/show` reports Clef as `["vision", "decision"]` — no `completion` — and + * discovery stores that as `supportedEndpoints: ["systemone"]`. Before this guard the row + * reached `/v1/models` with no `type` (so agents listed it as a chat model) and a chat call + * went upstream and came back as a 400. + * + * Rules: + * R1 The endpoint policy classifies a systemone-only row as `decision`, not chat-selectable. + * R2 `/v1/models` classification tags a systemone-only row `type: "decision"`; a model that + * also advertises chat (Nimble, Tev) stays a chat model. + * R3 A chat call to a decision-only model is refused with a clear 400, before any upstream + * call, and the message is one a combo treats as model-scoped (advance, never stop). + * R4 Chat-capable, unknown, and non-System-One-provider models are untouched, and other + * providers never pay the stored-model lookup. + * R5 resolveModelOrError — the chokepoint for direct calls and every combo target — + * returns that 400 for a decision-only model. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-decision-guard-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = "test-api-key-secret-decision-guard"; + +const core = await import("../../src/lib/db/core.ts"); +const models = await import("../../src/lib/db/models.ts"); +const { getModelEndpointDecision, isChatSelectableModel } = + await import("../../open-sse/services/modelEndpointPolicy.ts"); +const { classifyModelSupportedEndpoints } = + await import("../../src/shared/constants/modelSupportedEndpoints.ts"); +const { comboTargetDecision } = + await import("../../open-sse/services/combo/statusDecisionTable.ts"); +const { decisionOnlyChatRejection } = + await import("../../src/lib/providerModels/decisionOnlyChatGuard.ts"); +const { resolveModelOrError } = await import("../../src/sse/handlers/chatHelpers.ts"); + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +await models.replaceSyncedAvailableModelsForConnection("ollama-local", "conn-1", [ + { id: "clef-flash", name: "clef-flash", supportedEndpoints: ["systemone"] }, + { id: "nimble:latest", name: "nimble:latest", supportedEndpoints: ["systemone", "chat"] }, + { id: "gemma3:4b", name: "gemma3:4b", supportedEndpoints: ["chat"] }, +]); + +test("R1: the endpoint policy classifies a systemone-only row as a decision model", () => { + const decision = getModelEndpointDecision("ollama-local", "clef-flash", ["systemone"]); + assert.equal(decision.kind, "decision"); + assert.equal(decision.chatSelectable, false); + assert.equal( + isChatSelectableModel("ollama-local", { id: "clef", supportedEndpoints: ["systemone"] }), + false + ); + + const both = getModelEndpointDecision("ollama-local", "nimble", ["systemone", "chat"]); + assert.equal(both.kind, "chat"); + assert.equal(both.chatSelectable, true); +}); + +test("R2: /v1/models tags a systemone-only row as decision; chat + systemone stays chat", () => { + assert.deepEqual(classifyModelSupportedEndpoints(["systemone"]), { type: "decision" }); + assert.deepEqual(classifyModelSupportedEndpoints(["systemone", "chat"]), {}); + assert.deepEqual(classifyModelSupportedEndpoints(["chat"]), {}); +}); + +test("R3: a chat call to a decision-only model is a clear, combo-advancing 400", async () => { + const response = await decisionOnlyChatRejection({ + provider: "ollama-local", + model: "clef-flash", + }); + assert.ok(response, "a decision-only model must be refused"); + assert.equal(response.status, 400); + const body = await response.json(); + const message = String(body.error?.message ?? ""); + assert.match(message, /does not support chat/i); + assert.match(message, /System One API \(POST v1\/systemone\)/); + assert.equal(message.includes("at /"), false, "no stack trace in the body"); + assert.equal( + comboTargetDecision(400, message), + "advance", + "a combo must move to the next target" + ); +}); + +test("R4: chat-capable, unknown and other-provider models are untouched", async () => { + assert.equal( + await decisionOnlyChatRejection({ provider: "ollama-local", model: "nimble:latest" }), + null + ); + assert.equal( + await decisionOnlyChatRejection({ provider: "ollama-local", model: "gemma3:4b" }), + null + ); + assert.equal( + await decisionOnlyChatRejection({ provider: "ollama-local", model: "not-synced" }), + null + ); + + let lookups = 0; + const spy = async () => { + lookups += 1; + return []; + }; + assert.equal( + await decisionOnlyChatRejection( + { provider: "openai", model: "gpt-5.5" }, + { getSyncedAvailableModels: spy } + ), + null + ); + assert.equal(lookups, 0, "providers that cannot serve System One never pay the lookup"); +}); + +test("R5: resolveModelOrError refuses a decision-only model before any dispatch", async () => { + const body = { model: "ollama-local/clef-flash", messages: [{ role: "user", content: "hi" }] }; + const resolved = (await resolveModelOrError( + "ollama-local/clef-flash", + body, + "/v1/chat/completions" + )) as { + error?: Response; + }; + assert.ok(resolved.error, "a decision-only model must not resolve for chat"); + assert.equal(resolved.error.status, 400); + + const chat = (await resolveModelOrError( + "ollama-local/gemma3:4b", + body, + "/v1/chat/completions" + )) as { + error?: Response; + }; + assert.equal(chat.error, undefined, "a chat model still resolves"); +}); diff --git a/tests/unit/ollama-systemone.test.ts b/tests/unit/ollama-systemone.test.ts new file mode 100644 index 00000000000..d6135704cc0 --- /dev/null +++ b/tests/unit/ollama-systemone.test.ts @@ -0,0 +1,511 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-ollama-systemone-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const { applyOllamaShowCapabilities } = + await import("../../src/lib/providerModels/ollamaCapabilities.ts"); +const systemOne = await import("../../open-sse/handlers/ollamaSystemOne.ts"); +const { + buildOllamaSystemOneUrl, + classifyOllamaSystemOneFailure, + handleOllamaSystemOne, + validateOllamaSystemOneRequest, +} = systemOne; + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +// Captured from Ollama 0.35.0 (`/api/show` for nimble:latest and tev1:4b, 2026-09-30). +const SYSTEM_ONE_SHOW = { capabilities: ["decision", "tools", "thinking", "completion"] }; + +// Captured from Ollama 0.35.0 `POST /v1/systemone` with tev1:4b (2026-09-30). +const TEV_RESPONSE = { + model: "tev1:4b", + answers: { + eligible: { type: "noul", noul: 0.012264916251606712 }, + route: { + type: "choice", + choice: "repairs", + probabilities: { + refunds: 0.0431434161733377, + repairs: 0.946847119757787, + sales: 0.010009464068875148, + }, + confidence: 0.7875411973870763, + }, + urgency: { + type: "score", + score: 1.4291681505763034, + legend: { "0": "low", "1": "medium", "2": "high" }, + probabilities: { + "0": 0.10908117502705174, + "1": 0.35266949936959296, + "2": 0.5382493256033553, + }, + confidence: 0.14195633500571092, + }, + }, + usage: { input_tokens: 817, output_tokens: 4 }, +}; + +const VALID_BODY = { + model: "tev1:4b", + state: { ticket: "Laptop bought 45 days ago, screen flickers, wants a refund." }, + questions: { + eligible: { type: "noul", instructions: "Is the customer eligible for a full refund?" }, + route: { + type: "choice", + instructions: "Which team should handle this ticket?", + criteria: { refunds: "Refunds", repairs: "Repairs", sales: "Sales" }, + }, + urgency: { + type: "score", + instructions: "How urgent is this ticket?", + criteria: ["low", "medium", "high"], + }, + }, +}; + +function validBody(overrides: Record = {}) { + const result = validateOllamaSystemOneRequest({ ...VALID_BODY, ...overrides }); + assert.equal(result.ok, true); + if (!result.ok) throw new Error("unreachable"); + return result.data; +} + +type FetchCall = { url: string; init: RequestInit }; + +function jsonFetch(status: number, payload: unknown, calls: FetchCall[] = []): typeof fetch { + return (async (url: string | URL | Request, init?: RequestInit) => { + calls.push({ url: String(url), init: init ?? {} }); + return new Response(typeof payload === "string" ? payload : JSON.stringify(payload), { + status, + headers: { "Content-Type": "application/json" }, + }); + }) as typeof fetch; +} + +function recorder() { + const marks: unknown[][] = []; + const logs: Record[] = []; + const clears: unknown[] = []; + return { + marks, + logs, + clears, + deps: { + markAccountUnavailable: async (...args: unknown[]) => { + marks.push(args); + }, + logCall: async (entry: Record) => { + logs.push(entry); + }, + clearRecoveredState: async (credentials: unknown) => { + clears.push(credentials); + }, + }, + }; +} + +const CREDENTIALS = { + connectionId: "conn-1", + providerSpecificData: { baseUrl: "http://jetson.local:11434/v1" }, +}; + +// ── Capability discovery ──────────────────────────────────────────────────── + +test("Ollama `decision` capability advertises the systemone endpoint alongside chat", () => { + const model = applyOllamaShowCapabilities({ id: "nimble:latest" }, SYSTEM_ONE_SHOW); + assert.deepEqual(model.supportedEndpoints, ["systemone", "chat"]); + assert.equal(model.apiFormat, "chat-completions"); + assert.equal(model.supportsTools, true); + assert.equal(model.supportsThinking, true); +}); + +test("a decision-only model keeps its existing apiFormat instead of becoming an image model", () => { + const model = applyOllamaShowCapabilities( + { id: "decider", apiFormat: "chat-completions" }, + { capabilities: ["decision"] } + ); + assert.deepEqual(model.supportedEndpoints, ["systemone"]); + assert.equal(model.apiFormat, "chat-completions"); + + const bare = applyOllamaShowCapabilities({ id: "decider" }, { capabilities: ["decision"] }); + assert.equal("apiFormat" in bare, false); +}); + +test("ordinary chat models are unchanged by the decision mapping", () => { + const model = applyOllamaShowCapabilities( + { id: "gemma3:4b" }, + { capabilities: ["completion", "vision"] } + ); + assert.deepEqual(model.supportedEndpoints, ["chat"]); + assert.equal(model.supportsVision, true); +}); + +// ── Request validation (mirrors Ollama 0.35.0's own responses) ────────────── + +test("accepts all three question types and object/array/empty-object state", () => { + validBody(); + validBody({ state: ["a", "b"] }); + validBody({ state: {} }); + validBody({ keep_alive: "10m" }); + validBody({ keep_alive: 0 }); +}); + +test("drops unknown top-level fields such as stream, as Ollama ignores them", () => { + const data = validBody({ stream: true, foo: 1 }); + assert.equal("stream" in data, false); + assert.equal("foo" in data, false); +}); + +test("rejects the same bodies Ollama rejects, with a 400", () => { + const cases: Array<[string, Record, RegExp]> = [ + ["whitespace state", { state: " " }, /state/], + ["numeric state", { state: 5 }, /state/], + ["no questions", { questions: {} }, /questions must contain 1–64 fields/], + [ + "choice with one option", + { questions: { a: { type: "choice", instructions: "x", criteria: { only: "one" } } } }, + /criteria must contain 2–26 candidates/, + ], + [ + "score with one level", + { questions: { a: { type: "score", instructions: "x", criteria: ["low"] } } }, + /criteria must contain 2–26 candidates/, + ], + ["unknown question type", { questions: { a: { type: "rank", instructions: "x" } } }, /type/], + ["missing instructions", { questions: { a: { type: "noul" } } }, /instructions/], + ["missing model", { model: "" }, /model/], + ]; + for (const [name, overrides, pattern] of cases) { + const result = validateOllamaSystemOneRequest({ ...VALID_BODY, ...overrides }); + assert.equal(result.ok, false, name); + if (result.ok) continue; + assert.equal(result.status, 400, name); + assert.match(result.message, pattern, name); + } +}); + +test("rejects more than 64 questions", () => { + const questions = Object.fromEntries( + Array.from({ length: 65 }, (_, i) => [`q${i}`, { type: "noul", instructions: "?" }]) + ); + const result = validateOllamaSystemOneRequest({ ...VALID_BODY, questions }); + assert.equal(result.ok, false); + if (!result.ok) assert.equal(result.status, 400); +}); + +test("rejects a body over 64 KiB with 413, like Ollama", () => { + const result = validateOllamaSystemOneRequest({ ...VALID_BODY, state: "x".repeat(70_000) }); + assert.equal(result.ok, false); + if (result.ok) return; + assert.equal(result.status, 413); + assert.match(result.message, /64 KiB/); +}); + +// ── URL + failure classification ──────────────────────────────────────────── + +test("builds the native systemone URL from any OpenAI-compatible base URL", () => { + assert.equal(buildOllamaSystemOneUrl("http://h:11434/v1"), "http://h:11434/v1/systemone"); + assert.equal(buildOllamaSystemOneUrl("http://h:11434/v1/"), "http://h:11434/v1/systemone"); + assert.equal(buildOllamaSystemOneUrl("http://h:11434"), "http://h:11434/v1/systemone"); + assert.equal( + buildOllamaSystemOneUrl("http://h:11434/v1/chat/completions"), + "http://h:11434/v1/systemone" + ); + assert.equal(buildOllamaSystemOneUrl(null), "http://localhost:11434/v1/systemone"); +}); + +test("classifies upstream failures into the right resilience layer", () => { + assert.equal( + classifyOllamaSystemOneFailure(404, 'model "nimble" not found, try pulling it first'), + "model_not_found" + ); + assert.equal( + classifyOllamaSystemOneFailure( + 400, + 'model "gemma3:4b" is not supported by System One; use a local Nimble or Tev GGUF model' + ), + "unsupported_model" + ); + assert.equal( + classifyOllamaSystemOneFailure(400, "questions must contain 1–64 fields"), + "invalid_request" + ); + assert.equal(classifyOllamaSystemOneFailure(500, "boom"), "upstream_error"); +}); + +// ── Handler ───────────────────────────────────────────────────────────────── + +test("forwards only known fields to the connection's host and echoes the routed model id", async () => { + const calls: FetchCall[] = []; + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody({ keep_alive: "5m", stream: true }), + requestedModel: "ollama-local/tev1:4b", + credentials: CREDENTIALS, + fetchImpl: jsonFetch(200, TEV_RESPONSE, calls), + ...rec.deps, + }); + + assert.equal(response.status, 200); + assert.equal(calls.length, 1); + assert.equal(calls[0].url, "http://jetson.local:11434/v1/systemone"); + const sent = JSON.parse(String(calls[0].init.body)); + assert.deepEqual(Object.keys(sent).sort(), ["keep_alive", "model", "questions", "state"]); + assert.equal(sent.model, "tev1:4b"); + + const body = await response.json(); + assert.equal(body.model, "ollama-local/tev1:4b"); + assert.deepEqual(body.answers, TEV_RESPONSE.answers); + assert.deepEqual(body.usage, TEV_RESPONSE.usage); + + assert.equal(rec.marks.length, 0); + assert.equal(rec.clears.length, 1); + assert.equal(rec.logs.length, 1); + assert.deepEqual(rec.logs[0].tokens, { prompt_tokens: 817, completion_tokens: 4 }); + assert.equal(rec.logs[0].path, "/v1/systemone"); + assert.equal(rec.logs[0].connectionId, "conn-1"); +}); + +test("a missing model (404) locks that model on the connection", async () => { + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody({ model: "nimble" }), + requestedModel: "ollama-local/nimble", + credentials: CREDENTIALS, + fetchImpl: jsonFetch(404, { error: 'model "nimble" not found, try pulling it first' }), + ...rec.deps, + }); + + assert.equal(response.status, 404); + const body = await response.json(); + assert.match(body.error.message, /not found/); + assert.equal(rec.marks.length, 1); + const [connectionId, status, , provider, model] = rec.marks[0]; + assert.deepEqual( + [connectionId, status, provider, model], + ["conn-1", 404, "ollama-local", "nimble"] + ); +}); + +test("a non-System-One model (400) is returned as-is without touching the connection", async () => { + const rec = recorder(); + const message = + 'model "gemma3:4b" is not supported by System One; use a local Nimble or Tev GGUF model'; + const response = await handleOllamaSystemOne({ + body: validBody({ model: "gemma3:4b" }), + requestedModel: "ollama-local/gemma3:4b", + credentials: CREDENTIALS, + fetchImpl: jsonFetch(400, { error: message }), + ...rec.deps, + }); + + assert.equal(response.status, 400); + const body = await response.json(); + assert.match(body.error.message, /not supported by System One/); + assert.equal(rec.marks.length, 0); +}); + +test("an unreachable host returns 503 and cools the connection, without leaking internals", async () => { + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody(), + requestedModel: "ollama-local/tev1:4b", + credentials: CREDENTIALS, + fetchImpl: (async () => { + throw new TypeError("fetch failed: connect ECONNREFUSED 10.0.0.5:11434"); + }) as typeof fetch, + ...rec.deps, + }); + + assert.equal(response.status, 503); + const body = await response.json(); + assert.equal(body.error.message.includes("at /"), false); + assert.equal(body.error.message.includes("ECONNREFUSED"), false); + assert.equal(rec.marks.length, 1); + assert.equal(rec.marks[0][1], 503); +}); + +function hangingFetch(): typeof fetch { + return ((_url: string | URL | Request, init?: RequestInit) => + new Promise((_resolve, reject) => { + init?.signal?.addEventListener("abort", () => reject(new Error("aborted")), { once: true }); + })) as typeof fetch; +} + +test("a slow upstream times out with 504 and cools the connection", async () => { + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody(), + requestedModel: "ollama-local/tev1:4b", + credentials: CREDENTIALS, + timeoutMs: 20, + fetchImpl: hangingFetch(), + ...rec.deps, + }); + + assert.equal(response.status, 504); + assert.equal(rec.marks.length, 1); + assert.equal(rec.marks[0][1], 504); +}); + +test("a client disconnect does not cool the connection", async () => { + const rec = recorder(); + const controller = new AbortController(); + const pending = handleOllamaSystemOne({ + body: validBody(), + requestedModel: "ollama-local/tev1:4b", + credentials: CREDENTIALS, + signal: controller.signal, + fetchImpl: hangingFetch(), + ...rec.deps, + }); + controller.abort(); + const response = await pending; + + assert.equal(response.status, 499); + assert.equal(rec.marks.length, 0); +}); + +test("a 200 without answers is treated as a 502 upstream error", async () => { + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody(), + requestedModel: "ollama-local/tev1:4b", + credentials: CREDENTIALS, + fetchImpl: jsonFetch(200, "not json"), + ...rec.deps, + }); + + assert.equal(response.status, 502); + assert.equal(rec.marks.length, 1); +}); + +test("without a connection id nothing is marked, and the default host is used", async () => { + const calls: FetchCall[] = []; + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody(), + requestedModel: "ollama-local/tev1:4b", + credentials: null, + fetchImpl: jsonFetch(500, { error: "boom" }, calls), + ...rec.deps, + }); + + assert.equal(response.status, 500); + assert.equal(calls[0].url, "http://localhost:11434/v1/systemone"); + assert.equal(rec.marks.length, 0); + assert.equal(rec.clears.length, 0); +}); + +// ── Real resilience wiring ────────────────────────────────────────────────── + +test("a 404 through the real markAccountUnavailable locks only that model", async () => { + const providersDb = await import("../../src/lib/db/providers.ts"); + const { isModelLocked } = await import("../../open-sse/services/accountFallback.ts"); + + const connection = await providersDb.createProviderConnection({ + provider: "ollama-local", + authType: "none", + baseUrl: "http://127.0.0.1:11434/v1", + isActive: true, + }); + assert.ok(connection); + const connectionId = String(connection.id); + + const response = await handleOllamaSystemOne({ + body: validBody({ model: "nimble" }), + requestedModel: "ollama-local/nimble", + credentials: { connectionId }, + fetchImpl: jsonFetch(404, { error: 'model "nimble" not found, try pulling it first' }), + logCall: async () => {}, + }); + + assert.equal(response.status, 404); + assert.equal(isModelLocked("ollama-local", connectionId, "nimble"), true); + assert.equal(isModelLocked("ollama-local", connectionId, "tev1:4b"), false); + const stored = await providersDb.getProviderConnectionById(connectionId); + assert.notEqual(stored?.testStatus, "unavailable"); +}); + +// ── Clef / Clef Flash: vision System One models (Ollama >= 0.35.1) ────────── +// ollama.com/library/clef and /clef-flash tag both models `vision` + `decision`; the +// System One API takes base64 `images` shared by all questions (URLs and data URLs are +// not supported) and allows 32 MiB bodies when images are present (docs.ollama.com/api/systemone). + +// 1×1 transparent PNG. +const PNG_B64 = + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYAAAAAYAAjCB0C8AAAAASUVORK5CYII="; + +test("Clef: a vision + decision model advertises systemone and vision, not chat", () => { + const model = applyOllamaShowCapabilities( + { id: "clef-flash:9b" }, + { + capabilities: ["vision", "decision"], + } + ); + assert.deepEqual(model.supportedEndpoints, ["systemone"]); + assert.equal(model.supportsVision, true); + assert.equal("apiFormat" in model, false); +}); + +test("Clef: base64 images are validated and forwarded to Ollama", async () => { + const calls: FetchCall[] = []; + const rec = recorder(); + const response = await handleOllamaSystemOne({ + body: validBody({ model: "clef-flash", images: [PNG_B64, PNG_B64] }), + requestedModel: "ollama-local/clef-flash", + credentials: CREDENTIALS, + fetchImpl: jsonFetch(200, { ...TEV_RESPONSE, model: "clef-flash" }, calls), + ...rec.deps, + }); + assert.equal(response.status, 200); + const sent = JSON.parse(String(calls[0].init.body)); + assert.deepEqual(sent.images, [PNG_B64, PNG_B64]); + assert.deepEqual(Object.keys(sent).sort(), ["images", "model", "questions", "state"]); +}); + +test("Clef: image URLs, data URLs and non-string images are rejected with 400", () => { + for (const images of [ + ["https://example.com/cat.png"], + ["http://example.com/cat.png"], + [`data:image/png;base64,${PNG_B64}`], + [""], + [42], + PNG_B64, + ]) { + const result = validateOllamaSystemOneRequest({ ...VALID_BODY, images }); + assert.equal(result.ok, false, `images=${JSON.stringify(images).slice(0, 40)} must fail`); + if (result.ok) continue; + assert.equal(result.status, 400); + } +}); + +test("Clef: a body with images may exceed 64 KiB, up to Ollama's 32 MiB limit", () => { + const largeImage = "A".repeat(2 * 1024 * 1024); // ~2 MiB of base64 + const accepted = validateOllamaSystemOneRequest({ ...VALID_BODY, images: [largeImage] }); + assert.equal(accepted.ok, true); + + const tooLarge = validateOllamaSystemOneRequest({ + ...VALID_BODY, + images: ["A".repeat(33 * 1024 * 1024)], + }); + assert.equal(tooLarge.ok, false); + if (tooLarge.ok) return; + assert.equal(tooLarge.status, 413); + assert.match(tooLarge.message, /32 MiB/); + + // Without images the 64 KiB limit still applies. + const textOnly = validateOllamaSystemOneRequest({ ...VALID_BODY, state: "x".repeat(70_000) }); + assert.equal(textOnly.ok, false); +});