From eafe26a0853481b8ff2592c5ea36a6fc2631e836 Mon Sep 17 00:00:00 2001 From: Lawson Darrow Date: Fri, 29 May 2026 11:24:25 -0400 Subject: [PATCH 1/3] feat(shared): multimodal-aware prompt token estimation (#2112) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend estimateChatMessageTokens with opt-in, model-parameterized image awareness. When a model id is passed, image_url/image parts contribute the model's per-image token count (from imageInputTokensByResolution, default 560) and file/audio/video parts a flat default — mirroring the cost tables without serializing the raw blob. Without a model id the estimate stays text-only, so billing-path callers can opt out and avoid double-counting against the separate imageInputCost in costs.ts. Adds unit tests for text-only, mixed text+image, image-only, multi-image, unknown-model fallback, and file/audio cases. Refs #2112 Co-Authored-By: Claude Opus 4.8 (1M context) --- packages/shared/src/token-estimate.spec.ts | 75 +++++++++++++ packages/shared/src/token-estimate.ts | 119 ++++++++++++++++++--- 2 files changed, 180 insertions(+), 14 deletions(-) diff --git a/packages/shared/src/token-estimate.spec.ts b/packages/shared/src/token-estimate.spec.ts index 4cd4f6d929..cbcbf95885 100644 --- a/packages/shared/src/token-estimate.spec.ts +++ b/packages/shared/src/token-estimate.spec.ts @@ -89,3 +89,78 @@ describe("estimateChatMessageTokens", () => { expect(estimateChatMessageTokens(messages)).toBe(3); }); }); + +describe("estimateChatMessageTokens (multimodal-aware, with modelId)", () => { + // A real model id whose provider declares imageInputTokensByResolution: { default: 560 } + const IMAGE_MODEL = "gemini-3.1-flash-image-preview"; + const TOKENS_PER_IMAGE = 560; + + it("counts image-only content at the model's per-image rate", () => { + const messages = [ + { + content: [ + { + type: "image_url", + image_url: { url: "data:image/png;base64,AAAA" }, + }, + ], + }, + ]; + expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe( + TOKENS_PER_IMAGE, + ); + }); + + it("adds text tokens and per-image tokens for mixed content", () => { + const messages = [ + { + content: [ + { type: "text", text: "Describe this picture" }, // 21 chars → 5 + { + type: "image_url", + image_url: { url: "data:image/png;base64,AAAA" }, + }, + ], + }, + ]; + // round(21/4)=5, + 560 + expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe( + 5 + TOKENS_PER_IMAGE, + ); + }); + + it("counts each image part when several are present", () => { + const messages = [ + { + content: [ + { type: "image_url", image_url: { url: "a" } }, + { type: "image_url", image_url: { url: "b" } }, + ], + }, + ]; + expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe( + 2 * TOKENS_PER_IMAGE, + ); + }); + + it("falls back to the default per-image rate for an unknown model id", () => { + const messages = [ + { content: [{ type: "image_url", image_url: { url: "x" } }] }, + ]; + // 560 is also the documented default fallback + expect(estimateChatMessageTokens(messages, "no-such-model-xyz")).toBe( + TOKENS_PER_IMAGE, + ); + }); + + it("applies a flat default for file/audio/video parts", () => { + const messages = [{ content: [{ type: "input_audio" }] }]; + expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(560); + }); + + it("stays text-only behavior when no images are present", () => { + const messages = [{ content: "Hello" }, { content: ", world!" }]; + // same as the no-modelId path: (5 + 8) / 4 = 3 + expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(3); + }); +}); diff --git a/packages/shared/src/token-estimate.ts b/packages/shared/src/token-estimate.ts index 20e590e149..b8cda68232 100644 --- a/packages/shared/src/token-estimate.ts +++ b/packages/shared/src/token-estimate.ts @@ -1,8 +1,28 @@ +import { models, type ModelDefinition } from "@llmgateway/models"; + const CHARS_PER_TOKEN = 4; +/** + * Fallback per-image token count when a model has no + * `imageInputTokensByResolution` table. Mirrors `LEGACY_TOKENS_PER_INPUT_IMAGE` + * in apps/gateway/src/lib/costs.ts so the estimate and the cost calculation + * agree on the same default. + */ +const DEFAULT_TOKENS_PER_IMAGE = 560; + +/** + * File / audio / video parts have no per-model token table yet. We charge a + * flat, deliberately rough estimate so the fallback prompt-token count isn't + * wildly low for these payloads, without serializing the raw blob (which would + * massively over-count via chars/4). Tuned to the same order of magnitude as a + * single image. + */ +const DEFAULT_TOKENS_PER_NON_TEXT_PART = 560; + interface ContentPart { type?: string; text?: string; + image_url?: unknown; } interface MessageLike { @@ -23,22 +43,79 @@ export function estimateTokensFromText( } /** - * Returns the rough text-token count for a chat message array. + * Resolve the per-image token count for a model id, mirroring the per-model + * `imageInputTokensByResolution` tables used for billing in costs.ts. We don't + * know the provider/region that will ultimately serve the request at estimate + * time, so we take the first provider mapping that declares a table and prefer + * its `default` resolution entry. Falls back to `DEFAULT_TOKENS_PER_IMAGE`. + */ +function tokensPerImageForModel(modelId: string): number { + const model = models.find((m) => m.id === modelId) as + | ModelDefinition + | undefined; + if (!model) { + return DEFAULT_TOKENS_PER_IMAGE; + } + for (const provider of model.providers) { + const byResolution = provider.imageInputTokensByResolution; + if (byResolution) { + return ( + byResolution.default ?? + Object.values(byResolution)[0] ?? + DEFAULT_TOKENS_PER_IMAGE + ); + } + } + return DEFAULT_TOKENS_PER_IMAGE; +} + +function isImagePart(part: ContentPart): boolean { + return ( + part.type === "image_url" || + part.type === "image" || + Boolean(part.image_url) + ); +} + +function isOtherNonTextPart(part: ContentPart): boolean { + return ( + part.type === "file" || + part.type === "input_file" || + part.type === "audio" || + part.type === "input_audio" || + part.type === "video" + ); +} + +/** + * Returns a rough prompt-token estimate for a chat message array. * - * Only text payload is counted. Multimodal parts (image_url, file, audio, - * video, etc.) are intentionally ignored — image input billing is computed - * separately from `imageInputTokensByResolution` in costs.ts, so counting - * the serialized blob here would double-count. Tool/audio/video inputs are - * also not yet modeled and would distort the estimate if included verbatim. + * Text payload is counted as chars/4. When a `modelId` is supplied, multimodal + * parts are also counted: each `image_url`/`image` part contributes the model's + * per-image token count (from `imageInputTokensByResolution`, defaulting to + * `DEFAULT_TOKENS_PER_IMAGE`), and file/audio/video parts contribute a flat + * `DEFAULT_TOKENS_PER_NON_TEXT_PART`. The raw image/file blob is never counted + * via chars/4 — that would both over-count the payload here and double-count + * against the separate `imageInputCost` in costs.ts. * - * TODO: add multimodal-aware estimation that mirrors the per-model image - * token tables. Tracked in https://github.com/theopenco/llmgateway/issues/2112. + * When no `modelId` is supplied the estimate stays text-only (the historical + * behavior), so callers on the billing path — where image input is priced + * separately — can opt out of multimodal counting and avoid double-counting. + * + * Tracked in https://github.com/theopenco/llmgateway/issues/2112. */ -export function estimateChatMessageTokens(messages: MessageLike[]): number { +export function estimateChatMessageTokens( + messages: MessageLike[], + modelId?: string, +): number { if (!messages || messages.length === 0) { return 0; } + const countMultimodal = modelId !== undefined; + const tokensPerImage = countMultimodal ? tokensPerImageForModel(modelId) : 0; + let totalLength = 0; + let nonTextTokens = 0; for (const message of messages) { const content = message.content; if (typeof content === "string") { @@ -47,14 +124,28 @@ export function estimateChatMessageTokens(messages: MessageLike[]): number { } if (Array.isArray(content)) { for (const part of content) { - if (part && typeof part === "object" && typeof part.text === "string") { + if (!part || typeof part !== "object") { + continue; + } + if (typeof part.text === "string") { totalLength += part.text.length; + continue; + } + if (!countMultimodal) { + continue; + } + if (isImagePart(part)) { + nonTextTokens += tokensPerImage; + } else if (isOtherNonTextPart(part)) { + nonTextTokens += DEFAULT_TOKENS_PER_NON_TEXT_PART; } } } } - if (totalLength === 0) { - return 0; - } - return Math.max(1, Math.round(totalLength / CHARS_PER_TOKEN)); + + const textTokens = + totalLength === 0 + ? 0 + : Math.max(1, Math.round(totalLength / CHARS_PER_TOKEN)); + return textTokens + nonTextTokens; } From 9a49f5908c5f8a0f28b5529ac0010726a1b579bc Mon Sep 17 00:00:00 2001 From: Lawson Darrow Date: Sun, 31 May 2026 22:38:09 -0400 Subject: [PATCH 2/3] feat(gateway): image-aware token estimate for auto-routing (#2112) --- apps/gateway/src/chat/chat.ts | 19 ++++++++++++++++--- apps/gateway/src/chat/tools/tokenizer.ts | 13 ++++++++----- 2 files changed, 24 insertions(+), 8 deletions(-) diff --git a/apps/gateway/src/chat/chat.ts b/apps/gateway/src/chat/chat.ts index 5d24fe70bf..7276175d92 100644 --- a/apps/gateway/src/chat/chat.ts +++ b/apps/gateway/src/chat/chat.ts @@ -1787,9 +1787,22 @@ chat.openapi(completions, async (c) => { (usedProvider === "llmgateway" && usedInternalModel === "auto") || usedInternalModel === "auto" ) { - // Reuse the prompt-token estimate computed earlier so auto-routing can - // react to large prompts when picking a model. - const estimatedInputTokens = routingPromptTokens; + // Auto-routing and the context-window check below should react to image + // payloads, not just text (issue #2112). Recompute the estimate with an + // image-aware count instead of reusing the text-only routingPromptTokens. + // requestedModel may be "auto" here, in which case no per-model image + // table is found and the shared default per-image token count is used. + // This is kept separate from routingPromptTokens, which stays text-only: + // that value backs the billing-fallback usage numbers and image input is + // priced separately via imageInputCost in costs.ts (counting images + // there would double count). + let estimatedInputTokens = 0; + if (messages && messages.length > 0) { + estimatedInputTokens = encodeChatMessages(messages, requestedModel); + } + if (tools && tools.length > 0) { + estimatedInputTokens += Math.round(JSON.stringify(tools).length / 4); + } // Estimate the full context needed based on the request let requiredContextSize = estimatedInputTokens; diff --git a/apps/gateway/src/chat/tools/tokenizer.ts b/apps/gateway/src/chat/tools/tokenizer.ts index beb3049c68..68d88353a4 100644 --- a/apps/gateway/src/chat/tools/tokenizer.ts +++ b/apps/gateway/src/chat/tools/tokenizer.ts @@ -21,10 +21,13 @@ export function messageContentToString( /** * Rough length-based prompt-token estimate for a chat message array. * - * Backed by the shared `estimateChatMessageTokens` helper, which only counts - * text and ignores multimodal parts (image_url, file, etc.). Image input - * billing is handled separately in costs.ts. + * Backed by the shared `estimateChatMessageTokens` helper. By default this + * counts text only and ignores multimodal parts (image_url, file, etc.) — + * image input is priced separately in costs.ts, so the billing-fallback + * callers must not include it here. Pass `modelId` to additionally count + * multimodal parts (using the model's per-image token table); this is used for + * routing decisions that should reflect large image payloads. See issue #2112. */ -export function encodeChatMessages(messages: any[]): number { - return estimateChatMessageTokens(messages); +export function encodeChatMessages(messages: any[], modelId?: string): number { + return estimateChatMessageTokens(messages, modelId); } From 41a4623ab6e3518d641769a028bf62c6eef2623b Mon Sep 17 00:00:00 2001 From: Lawson Darrow Date: Mon, 1 Jun 2026 12:04:14 -0400 Subject: [PATCH 3/3] feat(gateway): warn when multimodal token estimate uses a default (#2112) estimateChatMessageTokens now reports via an optional onFallback callback when it had to use a rough default: the model has no per-image token table, or there are file/audio/video parts which have no per-model data yet. encodeChatMessages logs these via logger.warn so the unknown cases can be collected and the per-model token data improved over time, per maintainer feedback. Adds unit tests for the fallback reporting. --- apps/gateway/src/chat/tools/tokenizer.ts | 21 ++++++- packages/shared/src/index.ts | 1 + packages/shared/src/token-estimate.spec.ts | 45 ++++++++++++++ packages/shared/src/token-estimate.ts | 70 +++++++++++++++++----- 4 files changed, 120 insertions(+), 17 deletions(-) diff --git a/apps/gateway/src/chat/tools/tokenizer.ts b/apps/gateway/src/chat/tools/tokenizer.ts index 68d88353a4..eb1fd49926 100644 --- a/apps/gateway/src/chat/tools/tokenizer.ts +++ b/apps/gateway/src/chat/tools/tokenizer.ts @@ -1,4 +1,8 @@ -import { estimateChatMessageTokens } from "@llmgateway/shared"; +import { logger } from "@llmgateway/logger"; +import { + estimateChatMessageTokens, + type TokenEstimateFallback, +} from "@llmgateway/shared"; /** * Converts a message content value (string, array of content parts, null, or @@ -28,6 +32,19 @@ export function messageContentToString( * multimodal parts (using the model's per-image token table); this is used for * routing decisions that should reflect large image payloads. See issue #2112. */ +/** + * Logs when the multimodal estimate had to use a rough default, so unknown + * models/part types can be collected and the per-model token data improved + * over time (issue #2112). + */ +function warnOnFallbackEstimate(fallback: TokenEstimateFallback): void { + logger.warn("Multimodal token estimate fell back to a default", { + modelId: fallback.modelId, + imageParts: fallback.imageParts, + otherParts: fallback.otherParts, + }); +} + export function encodeChatMessages(messages: any[], modelId?: string): number { - return estimateChatMessageTokens(messages, modelId); + return estimateChatMessageTokens(messages, modelId, warnOnFallbackEstimate); } diff --git a/packages/shared/src/index.ts b/packages/shared/src/index.ts index f04a66174e..902cc94195 100644 --- a/packages/shared/src/index.ts +++ b/packages/shared/src/index.ts @@ -65,6 +65,7 @@ export { selectLoadBalancedItem } from "./load-balance.js"; export { estimateChatMessageTokens, estimateTokensFromText, + type TokenEstimateFallback, } from "./token-estimate.js"; export * from "./components/ui/index.js"; diff --git a/packages/shared/src/token-estimate.spec.ts b/packages/shared/src/token-estimate.spec.ts index cbcbf95885..37097bc72d 100644 --- a/packages/shared/src/token-estimate.spec.ts +++ b/packages/shared/src/token-estimate.spec.ts @@ -3,6 +3,7 @@ import { describe, it, expect } from "vitest"; import { estimateChatMessageTokens, estimateTokensFromText, + type TokenEstimateFallback, } from "./token-estimate.js"; describe("estimateTokensFromText", () => { @@ -164,3 +165,47 @@ describe("estimateChatMessageTokens (multimodal-aware, with modelId)", () => { expect(estimateChatMessageTokens(messages, IMAGE_MODEL)).toBe(3); }); }); + +describe("estimateChatMessageTokens (fallback reporting)", () => { + const IMAGE_MODEL = "gemini-3.1-flash-image-preview"; // has a per-image table + + it("reports image parts when the model has no per-image data", () => { + const messages = [ + { content: [{ type: "image_url", image_url: { url: "x" } }] }, + ]; + const calls: TokenEstimateFallback[] = []; + estimateChatMessageTokens(messages, "no-such-model-xyz", (f) => + calls.push(f), + ); + expect(calls).toEqual([ + { modelId: "no-such-model-xyz", imageParts: 1, otherParts: 0 }, + ]); + }); + + it("reports file/audio/video parts as fallbacks", () => { + const messages = [{ content: [{ type: "input_audio" }] }]; + const calls: TokenEstimateFallback[] = []; + estimateChatMessageTokens(messages, IMAGE_MODEL, (f) => calls.push(f)); + expect(calls).toEqual([ + { modelId: IMAGE_MODEL, imageParts: 0, otherParts: 1 }, + ]); + }); + + it("does not report when the model has per-image data", () => { + const messages = [ + { content: [{ type: "image_url", image_url: { url: "x" } }] }, + ]; + const calls: TokenEstimateFallback[] = []; + estimateChatMessageTokens(messages, IMAGE_MODEL, (f) => calls.push(f)); + expect(calls).toHaveLength(0); + }); + + it("does not report on the text-only path (no modelId)", () => { + const messages = [ + { content: [{ type: "image_url", image_url: { url: "x" } }] }, + ]; + const calls: TokenEstimateFallback[] = []; + estimateChatMessageTokens(messages, undefined, (f) => calls.push(f)); + expect(calls).toHaveLength(0); + }); +}); diff --git a/packages/shared/src/token-estimate.ts b/packages/shared/src/token-estimate.ts index b8cda68232..35b6dc018c 100644 --- a/packages/shared/src/token-estimate.ts +++ b/packages/shared/src/token-estimate.ts @@ -47,26 +47,41 @@ export function estimateTokensFromText( * `imageInputTokensByResolution` tables used for billing in costs.ts. We don't * know the provider/region that will ultimately serve the request at estimate * time, so we take the first provider mapping that declares a table and prefer - * its `default` resolution entry. Falls back to `DEFAULT_TOKENS_PER_IMAGE`. + * its `default` resolution entry. When the model has no table we fall back to + * `DEFAULT_TOKENS_PER_IMAGE` and report `isDefault` so callers can surface the + * unknown case. */ -function tokensPerImageForModel(modelId: string): number { +function resolveTokensPerImage(modelId: string): { + tokens: number; + isDefault: boolean; +} { const model = models.find((m) => m.id === modelId) as | ModelDefinition | undefined; - if (!model) { - return DEFAULT_TOKENS_PER_IMAGE; - } - for (const provider of model.providers) { - const byResolution = provider.imageInputTokensByResolution; - if (byResolution) { - return ( - byResolution.default ?? - Object.values(byResolution)[0] ?? - DEFAULT_TOKENS_PER_IMAGE - ); + if (model) { + for (const provider of model.providers) { + const byResolution = provider.imageInputTokensByResolution; + if (byResolution) { + const tokens = + byResolution.default ?? + Object.values(byResolution)[0] ?? + DEFAULT_TOKENS_PER_IMAGE; + return { tokens, isDefault: false }; + } } } - return DEFAULT_TOKENS_PER_IMAGE; + return { tokens: DEFAULT_TOKENS_PER_IMAGE, isDefault: true }; +} + +/** + * Reported when {@link estimateChatMessageTokens} had to fall back to a rough + * default because the model has no per-image table (`imageParts`) or because + * file/audio/video parts have no per-model token data yet (`otherParts`). + */ +export interface TokenEstimateFallback { + modelId: string; + imageParts: number; + otherParts: number; } function isImagePart(part: ContentPart): boolean { @@ -107,15 +122,24 @@ function isOtherNonTextPart(part: ContentPart): boolean { export function estimateChatMessageTokens( messages: MessageLike[], modelId?: string, + onFallback?: (fallback: TokenEstimateFallback) => void, ): number { if (!messages || messages.length === 0) { return 0; } const countMultimodal = modelId !== undefined; - const tokensPerImage = countMultimodal ? tokensPerImageForModel(modelId) : 0; + let tokensPerImage = 0; + let imageRateIsDefault = false; + if (countMultimodal) { + const resolved = resolveTokensPerImage(modelId); + tokensPerImage = resolved.tokens; + imageRateIsDefault = resolved.isDefault; + } let totalLength = 0; let nonTextTokens = 0; + let fallbackImageParts = 0; + let fallbackOtherParts = 0; for (const message of messages) { const content = message.content; if (typeof content === "string") { @@ -136,13 +160,29 @@ export function estimateChatMessageTokens( } if (isImagePart(part)) { nonTextTokens += tokensPerImage; + if (imageRateIsDefault) { + fallbackImageParts++; + } } else if (isOtherNonTextPart(part)) { nonTextTokens += DEFAULT_TOKENS_PER_NON_TEXT_PART; + fallbackOtherParts++; } } } } + if ( + countMultimodal && + onFallback && + (fallbackImageParts > 0 || fallbackOtherParts > 0) + ) { + onFallback({ + modelId: modelId as string, + imageParts: fallbackImageParts, + otherParts: fallbackOtherParts, + }); + } + const textTokens = totalLength === 0 ? 0