From beacd10e3374a92f478fc1e6f6045a7841c33712 Mon Sep 17 00:00:00 2001 From: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> Date: Sun, 26 Jul 2026 13:16:00 -0300 Subject: [PATCH 1/3] fix(sse): clamp max_tokens to the model output cap on every path enforceOutputTokenBudget only capped the three output-token fields against the remaining context window, so a request whose max_tokens exceeded the model's own output ceiling reached the upstream unchanged on the single-model path (the reasoning-token buffer covers only thinking models inside combo routing). Pass the model's explicit output cap into the budget check and use it as an extra upper bound when adjusting the fields. The reject decision stays tied to the context window: an output cap smaller than the default output budget must not turn a valid request into a 400. --- open-sse/handlers/chatCore.ts | 18 ++- .../handlers/chatCore/outputTokenBudget.ts | 27 ++++- .../output-token-budget-model-cap.test.ts | 108 ++++++++++++++++++ 3 files changed, 144 insertions(+), 9 deletions(-) create mode 100644 tests/unit/output-token-budget-model-cap.test.ts diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index af5c1ddfe27..17df6f8f47f 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -128,7 +128,12 @@ import { import { getUnsupportedParams, REGISTRY } from "../config/providerRegistry.ts"; import { stripUnsupportedParams } from "./chatCore/unsupportedParamsStrip.ts"; import { checkToolCallingRequiredButUnsupported } from "./chatCore/toolCallingRequiredCheck.ts"; -import { supportsMaxTokens, getResolvedModelCapabilities } from "@/lib/modelCapabilities.ts"; +import { + supportsMaxTokens, + getResolvedModelCapabilities, + getExplicitModelOutputCap, +} from "@/lib/modelCapabilities.ts"; +import { toPositiveInteger } from "../services/reasoningTokenBuffer.ts"; import { normalizeThinkingForModel } from "@/shared/constants/modelSpecs.ts"; import { buildErrorBody, @@ -1811,11 +1816,13 @@ export async function handleChatCore({ estimateTokens(body?.system) + estimateTokens(body?.instructions); const finalContextLimit = contextLimit; + const modelOutputCap = toPositiveInteger(getExplicitModelOutputCap(effectiveModel)); const outputBudget = enforceOutputTokenBudget( body as Record, finalEstimatedInputTokens, finalContextLimit, - targetFormat === FORMATS.CLAUDE && sourceFormat !== FORMATS.CLAUDE ? DEFAULT_MAX_TOKENS : 0 + targetFormat === FORMATS.CLAUDE && sourceFormat !== FORMATS.CLAUDE ? DEFAULT_MAX_TOKENS : 0, + modelOutputCap ); if (!outputBudget.ok) { const message = @@ -1833,10 +1840,15 @@ export async function handleChatCore({ ); } if (outputBudget.adjustedFields.length > 0) { + const cappedByModel = + modelOutputCap != null && modelOutputCap < outputBudget.availableOutputTokens; log?.info?.( "CONTEXT", `Adjusted invalid or oversized output token fields (${outputBudget.adjustedFields.join(", ")}); ` + - `${outputBudget.availableOutputTokens} tokens remain for output` + `${outputBudget.availableOutputTokens} tokens remain for output` + + (cappedByModel + ? ` (clamped to ${provider}/${effectiveModel}'s output cap of ${modelOutputCap})` + : "") ); } body = outputBudget.body; diff --git a/open-sse/handlers/chatCore/outputTokenBudget.ts b/open-sse/handlers/chatCore/outputTokenBudget.ts index 2d26d488c2a..0aaf008091f 100644 --- a/open-sse/handlers/chatCore/outputTokenBudget.ts +++ b/open-sse/handlers/chatCore/outputTokenBudget.ts @@ -22,12 +22,12 @@ type OutputTokenAdjustment = { field: string; value?: number; remove?: boolean } function getOutputTokenAdjustment( field: string, value: unknown, - availableOutputTokens: number + effectiveCap: number ): OutputTokenAdjustment | null { if (typeof value !== "number") return null; if (!Number.isFinite(value) || value <= 0) return { field, remove: true }; - const capped = Math.min(Math.floor(value), availableOutputTokens); + const capped = Math.min(Math.floor(value), effectiveCap); return capped === value ? null : { field, value: capped }; } @@ -40,10 +40,10 @@ function hasTranslatorOutputTokenLimit(body: Record): boolean { function adjustOutputTokenFields( body: Record, - availableOutputTokens: number + effectiveCap: number ): Pick, "body" | "adjustedFields"> { const adjustments = OUTPUT_TOKEN_FIELDS.map((field) => - getOutputTokenAdjustment(field, body[field], availableOutputTokens) + getOutputTokenAdjustment(field, body[field], effectiveCap) ).filter((adjustment): adjustment is OutputTokenAdjustment => adjustment !== null); if (adjustments.length === 0) return { body, adjustedFields: [] }; @@ -65,12 +65,22 @@ function adjustOutputTokenFields( * Reject that target locally instead of allowing the derived value to become * negative upstream. Positive client limits are capped to the remaining room; * invalid numeric limits are removed. + * + * `maxOutputTokenCap` (the model's own output ceiling, e.g. from + * `getExplicitModelOutputCap`) is an additional upper bound applied only when + * adjusting the output-token fields — never on the accept/reject decision, + * which stays tied to the context window alone. A model with a small output + * cap paired with a larger `defaultOutputTokens` must still be accepted; the + * cap limits how much is requested, not whether the request fits. Absent / + * null / non-positive cap values leave behavior byte-identical to before this + * parameter existed (fail-open). */ export function enforceOutputTokenBudget( body: Record | null | undefined, estimatedInputTokens: number, contextLimit: number, - defaultOutputTokens = 0 + defaultOutputTokens = 0, + maxOutputTokenCap?: number | null ): OutputTokenBudgetResult { const normalizedInputTokens = Math.max(0, Math.ceil(estimatedInputTokens)); const normalizedContextLimit = Math.max(1, Math.floor(contextLimit)); @@ -85,6 +95,11 @@ export function enforceOutputTokenBudget( }; } + const effectiveCap = + maxOutputTokenCap != null && maxOutputTokenCap > 0 + ? Math.min(availableOutputTokens, Math.floor(maxOutputTokenCap)) + : availableOutputTokens; + if (!body) { if (normalizedDefaultOutputTokens > availableOutputTokens) { return { @@ -112,6 +127,6 @@ export function enforceOutputTokenBudget( }; } - const adjusted = adjustOutputTokenFields(body, availableOutputTokens); + const adjusted = adjustOutputTokenFields(body, effectiveCap); return { ok: true, ...adjusted, availableOutputTokens }; } diff --git a/tests/unit/output-token-budget-model-cap.test.ts b/tests/unit/output-token-budget-model-cap.test.ts new file mode 100644 index 00000000000..25884f99096 --- /dev/null +++ b/tests/unit/output-token-budget-model-cap.test.ts @@ -0,0 +1,108 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { enforceOutputTokenBudget } from "../../open-sse/handlers/chatCore/outputTokenBudget.ts"; + +test("clamps max_tokens to the model output cap when the window has ample room", () => { + const result = enforceOutputTokenBudget({ max_tokens: 128_000 }, 1_000, 200_000, 0, 64_000); + + assert.equal(result.ok, true); + if (!result.ok) return; + assert.equal(result.body.max_tokens, 64_000); + assert.deepEqual(result.adjustedFields, ["max_tokens"]); +}); + +test("never elevates a max_tokens already below the model output cap", () => { + const result = enforceOutputTokenBudget({ max_tokens: 32_000 }, 1_000, 200_000, 0, 64_000); + + assert.equal(result.ok, true); + if (!result.ok) return; + assert.equal(result.body.max_tokens, 32_000); + assert.deepEqual(result.adjustedFields, []); +}); + +test("is byte-identical to the context-only behavior when the cap is absent", () => { + const withoutCapArg = enforceOutputTokenBudget({ max_tokens: 12_000 }, 127_000, 128_000); + const withUndefinedCap = enforceOutputTokenBudget( + { max_tokens: 12_000 }, + 127_000, + 128_000, + 0, + undefined + ); + const withNullCap = enforceOutputTokenBudget({ max_tokens: 12_000 }, 127_000, 128_000, 0, null); + + assert.deepEqual(withUndefinedCap, withoutCapArg); + assert.deepEqual(withNullCap, withoutCapArg); + assert.equal(withoutCapArg.ok, true); + if (!withoutCapArg.ok) return; + assert.equal(withoutCapArg.availableOutputTokens, 1_000); +}); + +test("clamps all three output-token field names to the model output cap", () => { + const result = enforceOutputTokenBudget( + { + max_tokens: 128_000, + max_completion_tokens: 128_000, + max_output_tokens: 128_000, + }, + 1_000, + 200_000, + 0, + 64_000 + ); + + assert.equal(result.ok, true); + if (!result.ok) return; + assert.equal(result.body.max_tokens, 64_000); + assert.equal(result.body.max_completion_tokens, 64_000); + assert.equal(result.body.max_output_tokens, 64_000); + assert.deepEqual( + result.adjustedFields.slice().sort(), + ["max_completion_tokens", "max_output_tokens", "max_tokens"].sort() + ); +}); + +test("the context window wins over the model output cap when the window is tighter", () => { + const result = enforceOutputTokenBudget({ max_tokens: 128_000 }, 127_000, 128_000, 0, 64_000); + + assert.equal(result.ok, true); + if (!result.ok) return; + // availableOutputTokens (1_000) < cap (64_000): the narrower window governs the clamp. + assert.equal(result.body.max_tokens, 1_000); + assert.equal(result.availableOutputTokens, 1_000); +}); + +test("a model output cap smaller than the default output budget does not reject the request", () => { + // Regression guard: the reject decision must stay tied to the context window only. + // A model with a small output ceiling (e.g. 4096) paired with a larger + // defaultOutputTokens must not turn a valid request into a 400. + const result = enforceOutputTokenBudget({}, 1_000, 200_000, 64_000, 4_096); + + assert.equal(result.ok, true); + if (!result.ok) return; + assert.equal(result.availableOutputTokens, 199_000); +}); + +test("adjustedFields reflects exactly the fields the model output cap changed", () => { + const result = enforceOutputTokenBudget( + { + max_tokens: 64_000, + max_completion_tokens: 128_000, + }, + 1_000, + 200_000, + 0, + 64_000 + ); + + assert.equal(result.ok, true); + if (!result.ok) return; + assert.equal( + result.body.max_tokens, + 64_000, + "already at the cap, must not be reported as adjusted" + ); + assert.equal(result.body.max_completion_tokens, 64_000); + assert.deepEqual(result.adjustedFields, ["max_completion_tokens"]); +}); From 7e7a9bba2a2319881ed0e34618fb766ef20ae746 Mon Sep 17 00:00:00 2001 From: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> Date: Sun, 26 Jul 2026 13:18:38 -0300 Subject: [PATCH 2/3] fix(sse): key the output-cap lookup by provider + model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The bare-string form of getExplicitModelOutputCap resolves to `provider: null`, which skips the registry cap and the operator's `max_token` capability override (#6524) — the documented escape hatch for a wrong synced `limit_output`. Clamping against a stale static spec while the operator had raised the ceiling would silently truncate output. Matches the { provider, model } form already used by the sibling capability lookups in this file (getResolvedModelCapabilities, supportsMaxTokens). --- open-sse/handlers/chatCore.ts | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 17df6f8f47f..c1c7c4ee574 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -1816,7 +1816,14 @@ export async function handleChatCore({ estimateTokens(body?.system) + estimateTokens(body?.instructions); const finalContextLimit = contextLimit; - const modelOutputCap = toPositiveInteger(getExplicitModelOutputCap(effectiveModel)); + // Key the lookup by { provider, model } — the bare-string form resolves to + // `provider: null`, which skips both the registry cap and the operator's + // `max_token` capability override (#6524), the documented escape hatch for a + // wrong synced `limit_output`. Clamping against a stale spec while the operator + // raised the ceiling would silently truncate output. + const modelOutputCap = toPositiveInteger( + getExplicitModelOutputCap({ provider, model: effectiveModel }) + ); const outputBudget = enforceOutputTokenBudget( body as Record, finalEstimatedInputTokens, From 28e7a22c8d55ed3a492e224c52b637451575eb85 Mon Sep 17 00:00:00 2001 From: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> Date: Sun, 26 Jul 2026 13:28:31 -0300 Subject: [PATCH 3/3] test(sse): cover the output-cap callsite; harden the sub-token cap guard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The unit tests drive enforceOutputTokenBudget() directly, so dropping the cap argument at the handleChatCore callsite left every one of them green. Add a wiring test that runs handleChatCore end to end against a stubbed fetch and asserts the body actually dispatched upstream. The cap comes from an operator `max_token` capability override rather than a catalog model: the override table is keyed by provider, so the test also pins the { provider, model } lookup — both the missing argument and the bare-string form fail it (verified by mutating each in turn). Also floor `maxOutputTokenCap` before the positivity test. A fractional cap below 1 previously passed `> 0` and floored to an effective cap of 0, clamping every field to zero; sub-token caps are meaningless and now read as absent. Unreachable through the callsite (toPositiveInteger filters it) but the exported contract was wrong. The adjustment log now states the output ceiling in effect instead of claiming the cap caused the adjustment — a field can also be adjusted by removal of an invalid value, which the cap did not cause. --- open-sse/handlers/chatCore.ts | 9 +- .../handlers/chatCore/outputTokenBudget.ts | 8 +- .../chatcore-model-output-cap-wiring.test.ts | 117 ++++++++++++++++++ .../output-token-budget-model-cap.test.ts | 12 ++ 4 files changed, 141 insertions(+), 5 deletions(-) create mode 100644 tests/unit/chatcore-model-output-cap-wiring.test.ts diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index c1c7c4ee574..2601f58a491 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -1847,14 +1847,17 @@ export async function handleChatCore({ ); } if (outputBudget.adjustedFields.length > 0) { - const cappedByModel = + // A field can also be adjusted by *removal* (invalid/non-positive value), which + // the cap did not cause — so state the ceiling in effect rather than claiming + // the cap drove this particular adjustment. + const modelCapIsBinding = modelOutputCap != null && modelOutputCap < outputBudget.availableOutputTokens; log?.info?.( "CONTEXT", `Adjusted invalid or oversized output token fields (${outputBudget.adjustedFields.join(", ")}); ` + `${outputBudget.availableOutputTokens} tokens remain for output` + - (cappedByModel - ? ` (clamped to ${provider}/${effectiveModel}'s output cap of ${modelOutputCap})` + (modelCapIsBinding + ? ` (output ceiling in effect: ${modelOutputCap}, ${provider}/${effectiveModel}'s own cap)` : "") ); } diff --git a/open-sse/handlers/chatCore/outputTokenBudget.ts b/open-sse/handlers/chatCore/outputTokenBudget.ts index 0aaf008091f..62752e42f46 100644 --- a/open-sse/handlers/chatCore/outputTokenBudget.ts +++ b/open-sse/handlers/chatCore/outputTokenBudget.ts @@ -95,9 +95,13 @@ export function enforceOutputTokenBudget( }; } + // Floor before the positivity test: a fractional cap below 1 would otherwise + // survive the `> 0` guard and floor to an effective cap of 0, clamping every + // field to zero. Sub-token caps are meaningless — treat them as absent. + const normalizedOutputCap = maxOutputTokenCap == null ? null : Math.floor(maxOutputTokenCap); const effectiveCap = - maxOutputTokenCap != null && maxOutputTokenCap > 0 - ? Math.min(availableOutputTokens, Math.floor(maxOutputTokenCap)) + normalizedOutputCap !== null && normalizedOutputCap > 0 + ? Math.min(availableOutputTokens, normalizedOutputCap) : availableOutputTokens; if (!body) { diff --git a/tests/unit/chatcore-model-output-cap-wiring.test.ts b/tests/unit/chatcore-model-output-cap-wiring.test.ts new file mode 100644 index 00000000000..f50caa39582 --- /dev/null +++ b/tests/unit/chatcore-model-output-cap-wiring.test.ts @@ -0,0 +1,117 @@ +// Wiring guard for the model-output-cap clamp: output-token-budget-model-cap.test.ts +// exercises enforceOutputTokenBudget() directly and therefore cannot catch a +// callsite regression — drop the cap argument in handleChatCore and every one of +// those unit tests still passes. This test drives handleChatCore() end to end +// (stubbed fetch, temp DB) and asserts the body actually dispatched upstream. +// +// The cap is supplied through an operator `max_token` capability override +// (src/lib/db/modelCapabilityOverrides.ts, issue #6524) rather than a catalog +// model, which pins two things at once and keeps the test independent of +// provider-catalog drift: +// 1. the clamp runs on the single-model (non-combo) path; +// 2. the cap lookup is keyed by { provider, model } — the bare-string form +// resolves to provider: null, and the override table is keyed by provider, +// so a string-keyed lookup silently misses it and no clamp happens. +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-model-output-cap-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const overridesDb = await import("../../src/lib/db/modelCapabilityOverrides.ts"); +const { handleChatCore } = await import("../../open-sse/handlers/chatCore.ts"); + +// Distinctive enough that it can never collide with a provider registered in +// open-sse/config/providerRegistry.ts, so nothing but the override supplies a cap. +const PROVIDER = "capwire-testprov"; +const MODEL = "capwire-testmodel"; +const OUTPUT_CAP = 1000; +const REQUESTED_MAX_TOKENS = 50_000; + +const originalFetch = globalThis.fetch; +let dispatchedBody: Record | null = null; + +const silentLog = { debug() {}, info() {}, warn() {}, error() {} }; + +function buildRequest(maxTokens: number) { + const body = { + model: MODEL, + messages: [{ role: "user", content: "hello" }], + max_tokens: maxTokens, + stream: false, + }; + return { + body, + modelInfo: { provider: PROVIDER, model: MODEL, extendedContext: false }, + credentials: { + apiKey: "sk-test", + providerSpecificData: { baseUrl: "https://capwire.example.test" }, + }, + clientRawRequest: { + endpoint: "/v1/chat/completions", + body, + headers: new Headers({ accept: "application/json" }), + }, + userAgent: "unit-test", + isCombo: false, + log: silentLog, + }; +} + +test.before(() => { + core.resetDbInstance(); + assert.equal( + overridesDb.setModelCapabilityOverride(`${PROVIDER}/${MODEL}`, "max_token", OUTPUT_CAP), + true, + "the operator override must be persisted for this test to mean anything" + ); + + globalThis.fetch = async (_input: RequestInfo | URL, init?: RequestInit) => { + dispatchedBody = init?.body ? JSON.parse(String(init.body)) : null; + return new Response( + JSON.stringify({ + id: "chatcmpl-capwire", + choices: [ + { index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { status: 200, headers: { "content-type": "application/json" } } + ); + }; +}); + +test.after(() => { + globalThis.fetch = originalFetch; + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("handleChatCore clamps an over-cap max_tokens to the model's output cap before dispatch", async () => { + dispatchedBody = null; + await handleChatCore(buildRequest(REQUESTED_MAX_TOKENS)); + + assert.ok(dispatchedBody, "expected the request to reach the upstream fetch"); + assert.equal( + dispatchedBody?.max_tokens, + OUTPUT_CAP, + `expected max_tokens clamped to the ${OUTPUT_CAP}-token operator cap, got ${dispatchedBody?.max_tokens}` + ); +}); + +test("handleChatCore leaves a max_tokens below the cap untouched", async () => { + dispatchedBody = null; + const underCap = OUTPUT_CAP - 1; + await handleChatCore(buildRequest(underCap)); + + assert.ok(dispatchedBody, "expected the request to reach the upstream fetch"); + assert.equal( + dispatchedBody?.max_tokens, + underCap, + "a request below the cap must never be raised to it" + ); +}); diff --git a/tests/unit/output-token-budget-model-cap.test.ts b/tests/unit/output-token-budget-model-cap.test.ts index 25884f99096..dd7dc42b17c 100644 --- a/tests/unit/output-token-budget-model-cap.test.ts +++ b/tests/unit/output-token-budget-model-cap.test.ts @@ -106,3 +106,15 @@ test("adjustedFields reflects exactly the fields the model output cap changed", assert.equal(result.body.max_completion_tokens, 64_000); assert.deepEqual(result.adjustedFields, ["max_completion_tokens"]); }); + +test("a sub-token cap is treated as absent, never as a cap of zero", () => { + // A fractional cap below 1 must not survive the positivity guard and floor to + // an effective cap of 0 — that would clamp every field to zero and either send + // `max_tokens: 0` upstream or bounce back through the translator default. + const result = enforceOutputTokenBudget({ max_tokens: 8_000 }, 1_000, 200_000, 0, 0.4); + + assert.equal(result.ok, true); + if (!result.ok) return; + assert.equal(result.body.max_tokens, 8_000, "sub-token cap must leave the request untouched"); + assert.deepEqual(result.adjustedFields, []); +});