Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 16 additions & 3 deletions apps/gateway/src/chat/chat.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1915,9 +1915,22 @@ chat.openapi(completions, async (c) => {
(usedProvider === "llmgateway" && usedInternalModel === "auto") ||
usedInternalModel === "auto"
) {
// Reuse the prompt-token estimate computed earlier so auto-routing can
// react to large prompts when picking a model.
const estimatedInputTokens = routingPromptTokens;
// Auto-routing and the context-window check below should react to image
// payloads, not just text (issue #2112). Recompute the estimate with an
// image-aware count instead of reusing the text-only routingPromptTokens.
// requestedModel may be "auto" here, in which case no per-model image
// table is found and the shared default per-image token count is used.
// This is kept separate from routingPromptTokens, which stays text-only:
// that value backs the billing-fallback usage numbers and image input is
// priced separately via imageInputCost in costs.ts (counting images
// there would double count).
let estimatedInputTokens = 0;
if (messages && messages.length > 0) {
estimatedInputTokens = encodeChatMessages(messages, requestedModel);
}
if (tools && tools.length > 0) {
estimatedInputTokens += Math.round(JSON.stringify(tools).length / 4);
}

// Estimate the full context needed based on the request
let requiredContextSize = estimatedInputTokens;
Expand Down
32 changes: 26 additions & 6 deletions apps/gateway/src/chat/tools/tokenizer.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,8 @@
import { estimateChatMessageTokens } from "@llmgateway/shared";
import { logger } from "@llmgateway/logger";
import {
estimateChatMessageTokens,
type TokenEstimateFallback,
} from "@llmgateway/shared";

/**
* Converts a message content value (string, array of content parts, null, or
Expand All @@ -21,10 +25,26 @@ export function messageContentToString(
/**
* Rough length-based prompt-token estimate for a chat message array.
*
* Backed by the shared `estimateChatMessageTokens` helper, which only counts
* text and ignores multimodal parts (image_url, file, etc.). Image input
* billing is handled separately in costs.ts.
* Backed by the shared `estimateChatMessageTokens` helper. By default this
* counts text only and ignores multimodal parts (image_url, file, etc.) —
* image input is priced separately in costs.ts, so the billing-fallback
* callers must not include it here. Pass `modelId` to additionally count
* multimodal parts (using the model's per-image token table); this is used for
* routing decisions that should reflect large image payloads. See issue #2112.
*/
export function encodeChatMessages(messages: any[]): number {
return estimateChatMessageTokens(messages);
/**
* Logs when the multimodal estimate had to use a rough default, so unknown
* models/part types can be collected and the per-model token data improved
* over time (issue #2112).
*/
function warnOnFallbackEstimate(fallback: TokenEstimateFallback): void {
logger.warn("Multimodal token estimate fell back to a default", {
modelId: fallback.modelId,
imageParts: fallback.imageParts,
otherParts: fallback.otherParts,
});
}

export function encodeChatMessages(messages: any[], modelId?: string): number {
return estimateChatMessageTokens(messages, modelId, warnOnFallbackEstimate);
}
1 change: 1 addition & 0 deletions packages/shared/src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,7 @@ export { selectLoadBalancedItem } from "./load-balance.js";
export {
estimateChatMessageTokens,
estimateTokensFromText,
type TokenEstimateFallback,
} from "./token-estimate.js";

export {
Expand Down
120 changes: 120 additions & 0 deletions packages/shared/src/token-estimate.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ import { describe, it, expect } from "vitest";
import {
estimateChatMessageTokens,
estimateTokensFromText,
type TokenEstimateFallback,
} from "./token-estimate.js";

describe("estimateTokensFromText", () => {
Expand Down Expand Up @@ -89,3 +90,122 @@ describe("estimateChatMessageTokens", () => {
expect(estimateChatMessageTokens(messages)).toBe(3);
});
});

describe("estimateChatMessageTokens (multimodal-aware, with modelId)", () => {
// A real model id whose provider declares imageInputTokensByResolution: { default: 560 }
const IMAGE_MODEL = "gemini-3.1-flash-image-preview";
const TOKENS_PER_IMAGE = 560;

it("counts image-only content at the model's per-image rate", () => {
const messages = [
{
content: [
{
type: "image_url",
image_url: { url: "data:image/png;base64,AAAA" },
},
],
},
];
expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(
TOKENS_PER_IMAGE,
);
});

it("adds text tokens and per-image tokens for mixed content", () => {
const messages = [
{
content: [
{ type: "text", text: "Describe this picture" }, // 21 chars → 5
{
type: "image_url",
image_url: { url: "data:image/png;base64,AAAA" },
},
],
},
];
// round(21/4)=5, + 560
expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(
5 + TOKENS_PER_IMAGE,
);
});

it("counts each image part when several are present", () => {
const messages = [
{
content: [
{ type: "image_url", image_url: { url: "a" } },
{ type: "image_url", image_url: { url: "b" } },
],
},
];
expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(
2 * TOKENS_PER_IMAGE,
);
});

it("falls back to the default per-image rate for an unknown model id", () => {
const messages = [
{ content: [{ type: "image_url", image_url: { url: "x" } }] },
];
// 560 is also the documented default fallback
expect(estimateChatMessageTokens(messages, "no-such-model-xyz")).toBe(
TOKENS_PER_IMAGE,
);
});

it("applies a flat default for file/audio/video parts", () => {
const messages = [{ content: [{ type: "input_audio" }] }];
expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(560);
});

it("stays text-only behavior when no images are present", () => {
const messages = [{ content: "Hello" }, { content: ", world!" }];
// same as the no-modelId path: (5 + 8) / 4 = 3
expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(3);
});
});

describe("estimateChatMessageTokens (fallback reporting)", () => {
const IMAGE_MODEL = "gemini-3.1-flash-image-preview"; // has a per-image table

it("reports image parts when the model has no per-image data", () => {
const messages = [
{ content: [{ type: "image_url", image_url: { url: "x" } }] },
];
const calls: TokenEstimateFallback[] = [];
estimateChatMessageTokens(messages, "no-such-model-xyz", (f) =>
calls.push(f),
);
expect(calls).toEqual([
{ modelId: "no-such-model-xyz", imageParts: 1, otherParts: 0 },
]);
});

it("reports file/audio/video parts as fallbacks", () => {
const messages = [{ content: [{ type: "input_audio" }] }];
const calls: TokenEstimateFallback[] = [];
estimateChatMessageTokens(messages, IMAGE_MODEL, (f) => calls.push(f));
expect(calls).toEqual([
{ modelId: IMAGE_MODEL, imageParts: 0, otherParts: 1 },
]);
});

it("does not report when the model has per-image data", () => {
const messages = [
{ content: [{ type: "image_url", image_url: { url: "x" } }] },
];
const calls: TokenEstimateFallback[] = [];
estimateChatMessageTokens(messages, IMAGE_MODEL, (f) => calls.push(f));
expect(calls).toHaveLength(0);
});

it("does not report on the text-only path (no modelId)", () => {
const messages = [
{ content: [{ type: "image_url", image_url: { url: "x" } }] },
];
const calls: TokenEstimateFallback[] = [];
estimateChatMessageTokens(messages, undefined, (f) => calls.push(f));
expect(calls).toHaveLength(0);
});
});
157 changes: 144 additions & 13 deletions packages/shared/src/token-estimate.ts
Original file line number Diff line number Diff line change
@@ -1,8 +1,28 @@
import { models, type ModelDefinition } from "@llmgateway/models";

const CHARS_PER_TOKEN = 4;

/**
* Fallback per-image token count when a model has no
* `imageInputTokensByResolution` table. Mirrors `LEGACY_TOKENS_PER_INPUT_IMAGE`
* in apps/gateway/src/lib/costs.ts so the estimate and the cost calculation
* agree on the same default.
*/
const DEFAULT_TOKENS_PER_IMAGE = 560;

/**
* File / audio / video parts have no per-model token table yet. We charge a
* flat, deliberately rough estimate so the fallback prompt-token count isn't
* wildly low for these payloads, without serializing the raw blob (which would
* massively over-count via chars/4). Tuned to the same order of magnitude as a
* single image.
*/
const DEFAULT_TOKENS_PER_NON_TEXT_PART = 560;

interface ContentPart {
type?: string;
text?: string;
image_url?: unknown;
}

interface MessageLike {
Expand All @@ -23,22 +43,103 @@ export function estimateTokensFromText(
}

/**
* Returns the rough text-token count for a chat message array.
* Resolve the per-image token count for a model id, mirroring the per-model
* `imageInputTokensByResolution` tables used for billing in costs.ts. We don't
* know the provider/region that will ultimately serve the request at estimate
* time, so we take the first provider mapping that declares a table and prefer
* its `default` resolution entry. When the model has no table we fall back to
* `DEFAULT_TOKENS_PER_IMAGE` and report `isDefault` so callers can surface the
* unknown case.
*/
function resolveTokensPerImage(modelId: string): {
tokens: number;
isDefault: boolean;
} {
const model = models.find((m) => m.id === modelId) as
| ModelDefinition
| undefined;
if (model) {
for (const provider of model.providers) {
const byResolution = provider.imageInputTokensByResolution;
if (byResolution) {
const tokens =
byResolution.default ??
Object.values(byResolution)[0] ??
DEFAULT_TOKENS_PER_IMAGE;
return { tokens, isDefault: false };
}
}
}
return { tokens: DEFAULT_TOKENS_PER_IMAGE, isDefault: true };
}

/**
* Reported when {@link estimateChatMessageTokens} had to fall back to a rough
* default because the model has no per-image table (`imageParts`) or because
* file/audio/video parts have no per-model token data yet (`otherParts`).
*/
export interface TokenEstimateFallback {
modelId: string;
imageParts: number;
otherParts: number;
}

function isImagePart(part: ContentPart): boolean {
return (
part.type === "image_url" ||
part.type === "image" ||
Boolean(part.image_url)
);
}

function isOtherNonTextPart(part: ContentPart): boolean {
return (
part.type === "file" ||
part.type === "input_file" ||
part.type === "audio" ||
part.type === "input_audio" ||
part.type === "video"
);
}

/**
* Returns a rough prompt-token estimate for a chat message array.
*
* Only text payload is counted. Multimodal parts (image_url, file, audio,
* video, etc.) are intentionally ignored — image input billing is computed
* separately from `imageInputTokensByResolution` in costs.ts, so counting
* the serialized blob here would double-count. Tool/audio/video inputs are
* also not yet modeled and would distort the estimate if included verbatim.
* Text payload is counted as chars/4. When a `modelId` is supplied, multimodal
* parts are also counted: each `image_url`/`image` part contributes the model's
* per-image token count (from `imageInputTokensByResolution`, defaulting to
* `DEFAULT_TOKENS_PER_IMAGE`), and file/audio/video parts contribute a flat
* `DEFAULT_TOKENS_PER_NON_TEXT_PART`. The raw image/file blob is never counted
* via chars/4 — that would both over-count the payload here and double-count
* against the separate `imageInputCost` in costs.ts.
*
* TODO: add multimodal-aware estimation that mirrors the per-model image
* token tables. Tracked in https://github.com/theopenco/llmgateway/issues/2112.
* When no `modelId` is supplied the estimate stays text-only (the historical
* behavior), so callers on the billing path — where image input is priced
* separately — can opt out of multimodal counting and avoid double-counting.
*
* Tracked in https://github.com/theopenco/llmgateway/issues/2112.
*/
export function estimateChatMessageTokens(messages: MessageLike[]): number {
export function estimateChatMessageTokens(
messages: MessageLike[],
modelId?: string,
onFallback?: (fallback: TokenEstimateFallback) => void,
): number {
if (!messages || messages.length === 0) {
return 0;
}
const countMultimodal = modelId !== undefined;
let tokensPerImage = 0;
let imageRateIsDefault = false;
if (countMultimodal) {
const resolved = resolveTokensPerImage(modelId);
tokensPerImage = resolved.tokens;
imageRateIsDefault = resolved.isDefault;
}

let totalLength = 0;
let nonTextTokens = 0;
let fallbackImageParts = 0;
let fallbackOtherParts = 0;
for (const message of messages) {
const content = message.content;
if (typeof content === "string") {
Expand All @@ -47,14 +148,44 @@ export function estimateChatMessageTokens(messages: MessageLike[]): number {
}
if (Array.isArray(content)) {
for (const part of content) {
if (part && typeof part === "object" && typeof part.text === "string") {
if (!part || typeof part !== "object") {
continue;
}
if (typeof part.text === "string") {
totalLength += part.text.length;
continue;
}
if (!countMultimodal) {
continue;
}
if (isImagePart(part)) {
nonTextTokens += tokensPerImage;
if (imageRateIsDefault) {
fallbackImageParts++;
}
} else if (isOtherNonTextPart(part)) {
nonTextTokens += DEFAULT_TOKENS_PER_NON_TEXT_PART;
fallbackOtherParts++;
}
}
}
}
if (totalLength === 0) {
return 0;

if (
countMultimodal &&
onFallback &&
(fallbackImageParts > 0 || fallbackOtherParts > 0)
) {
onFallback({
modelId: modelId as string,
imageParts: fallbackImageParts,
otherParts: fallbackOtherParts,
});
}
return Math.max(1, Math.round(totalLength / CHARS_PER_TOKEN));

const textTokens =
totalLength === 0
? 0
: Math.max(1, Math.round(totalLength / CHARS_PER_TOKEN));
return textTokens + nonTextTokens;
}
Loading