Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 25 additions & 3 deletions open-sse/handlers/chatCore.ts
Original file line number Diff line number Diff line change
Expand Up @@ -128,7 +128,12 @@ import {
import { getUnsupportedParams, REGISTRY } from "../config/providerRegistry.ts";
import { stripUnsupportedParams } from "./chatCore/unsupportedParamsStrip.ts";
import { checkToolCallingRequiredButUnsupported } from "./chatCore/toolCallingRequiredCheck.ts";
import { supportsMaxTokens, getResolvedModelCapabilities } from "@/lib/modelCapabilities.ts";
import {
supportsMaxTokens,
getResolvedModelCapabilities,
getExplicitModelOutputCap,
} from "@/lib/modelCapabilities.ts";
import { toPositiveInteger } from "../services/reasoningTokenBuffer.ts";
import { normalizeThinkingForModel } from "@/shared/constants/modelSpecs.ts";
import {
buildErrorBody,
Expand Down Expand Up @@ -1811,11 +1816,20 @@ export async function handleChatCore({
estimateTokens(body?.system) +
estimateTokens(body?.instructions);
const finalContextLimit = contextLimit;
// Key the lookup by { provider, model } — the bare-string form resolves to
// `provider: null`, which skips both the registry cap and the operator's
// `max_token` capability override (#6524), the documented escape hatch for a
// wrong synced `limit_output`. Clamping against a stale spec while the operator
// raised the ceiling would silently truncate output.
const modelOutputCap = toPositiveInteger(
getExplicitModelOutputCap({ provider, model: effectiveModel })
);
const outputBudget = enforceOutputTokenBudget(
body as Record<string, unknown>,
finalEstimatedInputTokens,
finalContextLimit,
targetFormat === FORMATS.CLAUDE && sourceFormat !== FORMATS.CLAUDE ? DEFAULT_MAX_TOKENS : 0
targetFormat === FORMATS.CLAUDE && sourceFormat !== FORMATS.CLAUDE ? DEFAULT_MAX_TOKENS : 0,
modelOutputCap
);
if (!outputBudget.ok) {
const message =
Expand All @@ -1833,10 +1847,18 @@ export async function handleChatCore({
);
}
if (outputBudget.adjustedFields.length > 0) {
// A field can also be adjusted by *removal* (invalid/non-positive value), which
// the cap did not cause — so state the ceiling in effect rather than claiming
// the cap drove this particular adjustment.
const modelCapIsBinding =
modelOutputCap != null && modelOutputCap < outputBudget.availableOutputTokens;
log?.info?.(
"CONTEXT",
`Adjusted invalid or oversized output token fields (${outputBudget.adjustedFields.join(", ")}); ` +
`${outputBudget.availableOutputTokens} tokens remain for output`
`${outputBudget.availableOutputTokens} tokens remain for output` +
(modelCapIsBinding
? ` (output ceiling in effect: ${modelOutputCap}, ${provider}/${effectiveModel}'s own cap)`
: "")
);
}
body = outputBudget.body;
Expand Down
31 changes: 25 additions & 6 deletions open-sse/handlers/chatCore/outputTokenBudget.ts
Original file line number Diff line number Diff line change
Expand Up @@ -22,12 +22,12 @@ type OutputTokenAdjustment = { field: string; value?: number; remove?: boolean }
function getOutputTokenAdjustment(
field: string,
value: unknown,
availableOutputTokens: number
effectiveCap: number
): OutputTokenAdjustment | null {
if (typeof value !== "number") return null;
if (!Number.isFinite(value) || value <= 0) return { field, remove: true };

const capped = Math.min(Math.floor(value), availableOutputTokens);
const capped = Math.min(Math.floor(value), effectiveCap);
return capped === value ? null : { field, value: capped };
}

Expand All @@ -40,10 +40,10 @@ function hasTranslatorOutputTokenLimit(body: Record<string, unknown>): boolean {

function adjustOutputTokenFields(
body: Record<string, unknown>,
availableOutputTokens: number
effectiveCap: number
): Pick<Extract<OutputTokenBudgetResult, { ok: true }>, "body" | "adjustedFields"> {
const adjustments = OUTPUT_TOKEN_FIELDS.map((field) =>
getOutputTokenAdjustment(field, body[field], availableOutputTokens)
getOutputTokenAdjustment(field, body[field], effectiveCap)
).filter((adjustment): adjustment is OutputTokenAdjustment => adjustment !== null);
if (adjustments.length === 0) return { body, adjustedFields: [] };

Expand All @@ -65,12 +65,22 @@ function adjustOutputTokenFields(
* Reject that target locally instead of allowing the derived value to become
* negative upstream. Positive client limits are capped to the remaining room;
* invalid numeric limits are removed.
*
* `maxOutputTokenCap` (the model's own output ceiling, e.g. from
* `getExplicitModelOutputCap`) is an additional upper bound applied only when
* adjusting the output-token fields — never on the accept/reject decision,
* which stays tied to the context window alone. A model with a small output
* cap paired with a larger `defaultOutputTokens` must still be accepted; the
* cap limits how much is requested, not whether the request fits. Absent /
* null / non-positive cap values leave behavior byte-identical to before this
* parameter existed (fail-open).
*/
export function enforceOutputTokenBudget(
body: Record<string, unknown> | null | undefined,
estimatedInputTokens: number,
contextLimit: number,
defaultOutputTokens = 0
defaultOutputTokens = 0,
maxOutputTokenCap?: number | null
): OutputTokenBudgetResult {
const normalizedInputTokens = Math.max(0, Math.ceil(estimatedInputTokens));
const normalizedContextLimit = Math.max(1, Math.floor(contextLimit));
Expand All @@ -85,6 +95,15 @@ export function enforceOutputTokenBudget(
};
}

// Floor before the positivity test: a fractional cap below 1 would otherwise
// survive the `> 0` guard and floor to an effective cap of 0, clamping every
// field to zero. Sub-token caps are meaningless — treat them as absent.
const normalizedOutputCap = maxOutputTokenCap == null ? null : Math.floor(maxOutputTokenCap);
const effectiveCap =
normalizedOutputCap !== null && normalizedOutputCap > 0
? Math.min(availableOutputTokens, normalizedOutputCap)
: availableOutputTokens;

if (!body) {
if (normalizedDefaultOutputTokens > availableOutputTokens) {
return {
Expand Down Expand Up @@ -112,6 +131,6 @@ export function enforceOutputTokenBudget(
};
}

const adjusted = adjustOutputTokenFields(body, availableOutputTokens);
const adjusted = adjustOutputTokenFields(body, effectiveCap);
return { ok: true, ...adjusted, availableOutputTokens };
}
117 changes: 117 additions & 0 deletions tests/unit/chatcore-model-output-cap-wiring.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,117 @@
// Wiring guard for the model-output-cap clamp: output-token-budget-model-cap.test.ts
// exercises enforceOutputTokenBudget() directly and therefore cannot catch a
// callsite regression — drop the cap argument in handleChatCore and every one of
// those unit tests still passes. This test drives handleChatCore() end to end
// (stubbed fetch, temp DB) and asserts the body actually dispatched upstream.
//
// The cap is supplied through an operator `max_token` capability override
// (src/lib/db/modelCapabilityOverrides.ts, issue #6524) rather than a catalog
// model, which pins two things at once and keeps the test independent of
// provider-catalog drift:
// 1. the clamp runs on the single-model (non-combo) path;
// 2. the cap lookup is keyed by { provider, model } — the bare-string form
// resolves to provider: null, and the override table is keyed by provider,
// so a string-keyed lookup silently misses it and no clamp happens.
import test from "node:test";
import assert from "node:assert/strict";
import fs from "node:fs";
import os from "node:os";
import path from "node:path";

const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-model-output-cap-"));
process.env.DATA_DIR = TEST_DATA_DIR;

const core = await import("../../src/lib/db/core.ts");
const overridesDb = await import("../../src/lib/db/modelCapabilityOverrides.ts");
const { handleChatCore } = await import("../../open-sse/handlers/chatCore.ts");

// Distinctive enough that it can never collide with a provider registered in
// open-sse/config/providerRegistry.ts, so nothing but the override supplies a cap.
const PROVIDER = "capwire-testprov";
const MODEL = "capwire-testmodel";
const OUTPUT_CAP = 1000;
const REQUESTED_MAX_TOKENS = 50_000;

const originalFetch = globalThis.fetch;
let dispatchedBody: Record<string, unknown> | null = null;

const silentLog = { debug() {}, info() {}, warn() {}, error() {} };

function buildRequest(maxTokens: number) {
const body = {
model: MODEL,
messages: [{ role: "user", content: "hello" }],
max_tokens: maxTokens,
stream: false,
};
return {
body,
modelInfo: { provider: PROVIDER, model: MODEL, extendedContext: false },
credentials: {
apiKey: "sk-test",
providerSpecificData: { baseUrl: "https://capwire.example.test" },
},
clientRawRequest: {
endpoint: "/v1/chat/completions",
body,
headers: new Headers({ accept: "application/json" }),
},
userAgent: "unit-test",
isCombo: false,
log: silentLog,
};
}

test.before(() => {
core.resetDbInstance();
assert.equal(
overridesDb.setModelCapabilityOverride(`${PROVIDER}/${MODEL}`, "max_token", OUTPUT_CAP),
true,
"the operator override must be persisted for this test to mean anything"
);

globalThis.fetch = async (_input: RequestInfo | URL, init?: RequestInit) => {
dispatchedBody = init?.body ? JSON.parse(String(init.body)) : null;
return new Response(
JSON.stringify({
id: "chatcmpl-capwire",
choices: [
{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" },
],
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
}),
{ status: 200, headers: { "content-type": "application/json" } }
);
};
});

test.after(() => {
globalThis.fetch = originalFetch;
core.resetDbInstance();
fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true });
});

test("handleChatCore clamps an over-cap max_tokens to the model's output cap before dispatch", async () => {
dispatchedBody = null;
await handleChatCore(buildRequest(REQUESTED_MAX_TOKENS));

assert.ok(dispatchedBody, "expected the request to reach the upstream fetch");
assert.equal(
dispatchedBody?.max_tokens,
OUTPUT_CAP,
`expected max_tokens clamped to the ${OUTPUT_CAP}-token operator cap, got ${dispatchedBody?.max_tokens}`
);
});

test("handleChatCore leaves a max_tokens below the cap untouched", async () => {
dispatchedBody = null;
const underCap = OUTPUT_CAP - 1;
await handleChatCore(buildRequest(underCap));

assert.ok(dispatchedBody, "expected the request to reach the upstream fetch");
assert.equal(
dispatchedBody?.max_tokens,
underCap,
"a request below the cap must never be raised to it"
);
});
120 changes: 120 additions & 0 deletions tests/unit/output-token-budget-model-cap.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,120 @@
import test from "node:test";
import assert from "node:assert/strict";

import { enforceOutputTokenBudget } from "../../open-sse/handlers/chatCore/outputTokenBudget.ts";

test("clamps max_tokens to the model output cap when the window has ample room", () => {
const result = enforceOutputTokenBudget({ max_tokens: 128_000 }, 1_000, 200_000, 0, 64_000);

assert.equal(result.ok, true);
if (!result.ok) return;
assert.equal(result.body.max_tokens, 64_000);
assert.deepEqual(result.adjustedFields, ["max_tokens"]);
});

test("never elevates a max_tokens already below the model output cap", () => {
const result = enforceOutputTokenBudget({ max_tokens: 32_000 }, 1_000, 200_000, 0, 64_000);

assert.equal(result.ok, true);
if (!result.ok) return;
assert.equal(result.body.max_tokens, 32_000);
assert.deepEqual(result.adjustedFields, []);
});

test("is byte-identical to the context-only behavior when the cap is absent", () => {
const withoutCapArg = enforceOutputTokenBudget({ max_tokens: 12_000 }, 127_000, 128_000);
const withUndefinedCap = enforceOutputTokenBudget(
{ max_tokens: 12_000 },
127_000,
128_000,
0,
undefined
);
const withNullCap = enforceOutputTokenBudget({ max_tokens: 12_000 }, 127_000, 128_000, 0, null);

assert.deepEqual(withUndefinedCap, withoutCapArg);
assert.deepEqual(withNullCap, withoutCapArg);
assert.equal(withoutCapArg.ok, true);
if (!withoutCapArg.ok) return;
assert.equal(withoutCapArg.availableOutputTokens, 1_000);
});

test("clamps all three output-token field names to the model output cap", () => {
const result = enforceOutputTokenBudget(
{
max_tokens: 128_000,
max_completion_tokens: 128_000,
max_output_tokens: 128_000,
},
1_000,
200_000,
0,
64_000
);

assert.equal(result.ok, true);
if (!result.ok) return;
assert.equal(result.body.max_tokens, 64_000);
assert.equal(result.body.max_completion_tokens, 64_000);
assert.equal(result.body.max_output_tokens, 64_000);
assert.deepEqual(
result.adjustedFields.slice().sort(),
["max_completion_tokens", "max_output_tokens", "max_tokens"].sort()
);
});

test("the context window wins over the model output cap when the window is tighter", () => {
const result = enforceOutputTokenBudget({ max_tokens: 128_000 }, 127_000, 128_000, 0, 64_000);

assert.equal(result.ok, true);
if (!result.ok) return;
// availableOutputTokens (1_000) < cap (64_000): the narrower window governs the clamp.
assert.equal(result.body.max_tokens, 1_000);
assert.equal(result.availableOutputTokens, 1_000);
});

test("a model output cap smaller than the default output budget does not reject the request", () => {
// Regression guard: the reject decision must stay tied to the context window only.
// A model with a small output ceiling (e.g. 4096) paired with a larger
// defaultOutputTokens must not turn a valid request into a 400.
const result = enforceOutputTokenBudget({}, 1_000, 200_000, 64_000, 4_096);

assert.equal(result.ok, true);
if (!result.ok) return;
assert.equal(result.availableOutputTokens, 199_000);
});

test("adjustedFields reflects exactly the fields the model output cap changed", () => {
const result = enforceOutputTokenBudget(
{
max_tokens: 64_000,
max_completion_tokens: 128_000,
},
1_000,
200_000,
0,
64_000
);

assert.equal(result.ok, true);
if (!result.ok) return;
assert.equal(
result.body.max_tokens,
64_000,
"already at the cap, must not be reported as adjusted"
);
assert.equal(result.body.max_completion_tokens, 64_000);
assert.deepEqual(result.adjustedFields, ["max_completion_tokens"]);
});

test("a sub-token cap is treated as absent, never as a cap of zero", () => {
// A fractional cap below 1 must not survive the positivity guard and floor to
// an effective cap of 0 — that would clamp every field to zero and either send
// `max_tokens: 0` upstream or bounce back through the translator default.
const result = enforceOutputTokenBudget({ max_tokens: 8_000 }, 1_000, 200_000, 0, 0.4);

assert.equal(result.ok, true);
if (!result.ok) return;
assert.equal(result.body.max_tokens, 8_000, "sub-token cap must leave the request untouched");
assert.deepEqual(result.adjustedFields, []);
});
Loading