From 9544119c34c9b4cf44338ca00b7e8053a5dade80 Mon Sep 17 00:00:00 2001 From: Hsia97 Date: Tue, 25 Aug 2026 10:36:24 +0800 Subject: [PATCH 1/3] fix(google): clamp max output tokens per model --- tests/google-output-clamp.test.ts | 33 +++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) create mode 100644 tests/google-output-clamp.test.ts diff --git a/tests/google-output-clamp.test.ts b/tests/google-output-clamp.test.ts new file mode 100644 index 00000000000..9f8f09718ae --- /dev/null +++ b/tests/google-output-clamp.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, test } from "bun:test"; +import { clampGoogleMaxOutputTokens, maxOutputTokensForGoogleModel } from "../src/adapters/google"; + +describe("google maxOutputTokens clamp", () => { + test("returns model-specific max output tokens limit", () => { + expect(maxOutputTokensForGoogleModel("gemini-3.7-flash")).toBe(65536); + expect(maxOutputTokensForGoogleModel("gemini-3.7-flash-tiered")).toBe(65536); + expect(maxOutputTokensForGoogleModel("gemini-3-pro")).toBe(65535); + expect(maxOutputTokensForGoogleModel("claude-3-7-sonnet")).toBe(64000); + expect(maxOutputTokensForGoogleModel("claude-3-5-sonnet@20241022")).toBe(64000); + expect(maxOutputTokensForGoogleModel("gpt-oss-120b")).toBe(32768); + expect(maxOutputTokensForGoogleModel("custom-unknown-model")).toBe(16384); + }); + + test("downward clamps excessive requested tokens to model max", () => { + expect(clampGoogleMaxOutputTokens("gemini-3.7-flash", 128000)).toBe(65536); + expect(clampGoogleMaxOutputTokens("gemini-3-pro", 100000)).toBe(65535); + expect(clampGoogleMaxOutputTokens("claude-3-7-sonnet", 100000)).toBe(64000); + expect(clampGoogleMaxOutputTokens("gpt-oss-120b", 64000)).toBe(32768); + }); + + test("preserves requested tokens when within model max", () => { + expect(clampGoogleMaxOutputTokens("gemini-3.7-flash", 4096)).toBe(4096); + expect(clampGoogleMaxOutputTokens("gemini-3-pro", 8192)).toBe(8192); + expect(clampGoogleMaxOutputTokens("claude-3-7-sonnet", 32000)).toBe(32000); + }); + + test("returns undefined when requested tokens is undefined or non-positive", () => { + expect(clampGoogleMaxOutputTokens("gemini-3.7-flash", undefined)).toBeUndefined(); + expect(clampGoogleMaxOutputTokens("gemini-3.7-flash", 0)).toBeUndefined(); + expect(clampGoogleMaxOutputTokens("gemini-3.7-flash", -10)).toBeUndefined(); + }); +}); From 8ac7d59bd1e3fc85b89bd549bfba194262c0f9e4 Mon Sep 17 00:00:00 2001 From: Hsia97 Date: Tue, 25 Aug 2026 10:36:36 +0800 Subject: [PATCH 2/3] fix(google): clamp max output tokens per model --- src/adapters/google.ts | 22 +++++++++++++++++++++- 1 file changed, 21 insertions(+), 1 deletion(-) diff --git a/src/adapters/google.ts b/src/adapters/google.ts index 19c142a9949..28fc890edec 100644 --- a/src/adapters/google.ts +++ b/src/adapters/google.ts @@ -48,6 +48,25 @@ const GOOGLE_BREVITY_INSTRUCTION = [ "- This applies only to intermediate progress text. Your final answer after the work is done is exempt: write it in full and at whatever length the task requires.", ].join("\n"); +export function maxOutputTokensForGoogleModel(modelId: string): number { + const lower = modelId.toLowerCase(); + if (lower.includes("flash")) return 65536; + if (lower.includes("pro")) return 65535; + if (lower.includes("claude")) return 64000; + if (lower.includes("gpt-oss") || lower.includes("oss")) return 32768; + if (lower.startsWith("gemini")) return 65536; + return 16384; +} + +export function clampGoogleMaxOutputTokens( + modelId: string, + requestedTokens?: number, +): number | undefined { + if (requestedTokens === undefined || requestedTokens <= 0) return undefined; + const modelMax = maxOutputTokensForGoogleModel(modelId); + return Math.min(requestedTokens, modelMax); +} + /** * Some Google direct deployments expose current Gemini Flash generations with a `-tiered` * wire suffix (`gemini-3.7-flash` -> `gemini-3.7-flash-tiered`). Keep the picker-visible id @@ -650,7 +669,8 @@ export function createGoogleAdapter(provider: OcxProviderConfig): ProviderAdapte if (toolConfig) body.toolConfig = toolConfig; const generationConfig: Record = {}; - if (parsed.options.maxOutputTokens) generationConfig.maxOutputTokens = parsed.options.maxOutputTokens; + const clampedMaxOutputTokens = clampGoogleMaxOutputTokens(identityModelId, parsed.options.maxOutputTokens); + if (clampedMaxOutputTokens !== undefined) generationConfig.maxOutputTokens = clampedMaxOutputTokens; if (parsed.options.temperature !== undefined) generationConfig.temperature = parsed.options.temperature; if (parsed.options.topP !== undefined) generationConfig.topP = parsed.options.topP; if (parsed.options.stopSequences) generationConfig.stopSequences = parsed.options.stopSequences; From c6d8edf6fb32eeac6087f57eb45d6fc98f49f691 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 26 Aug 2026 00:50:06 +0900 Subject: [PATCH 3/3] fix(google): do not invent an output ceiling for unrecognized models The clamp matched by substring and fell back to 16,384 for anything unmatched, so an alias, a gateway id, or any model newer than the table was silently truncated to 16,384 regardless of what the operator asked for. structure/02_config-and-codex-home.md is explicit that an explicit request value wins. Unknown ids now return undefined and pass the request through untouched; the upstream stays the authority on its own limit. Matching is also prefix/family based, because includes(pro) matched my-prototype-model and includes(oss) matched crossover-v2. --- src/adapters/google.ts | 34 +++++++++++++++++++++++-------- tests/google-output-clamp.test.ts | 17 +++++++++++++++- 2 files changed, 42 insertions(+), 9 deletions(-) diff --git a/src/adapters/google.ts b/src/adapters/google.ts index 28fc890edec..183b92615a2 100644 --- a/src/adapters/google.ts +++ b/src/adapters/google.ts @@ -48,14 +48,30 @@ const GOOGLE_BREVITY_INSTRUCTION = [ "- This applies only to intermediate progress text. Your final answer after the work is done is exempt: write it in full and at whatever length the task requires.", ].join("\n"); -export function maxOutputTokensForGoogleModel(modelId: string): number { - const lower = modelId.toLowerCase(); - if (lower.includes("flash")) return 65536; - if (lower.includes("pro")) return 65535; - if (lower.includes("claude")) return 64000; - if (lower.includes("gpt-oss") || lower.includes("oss")) return 32768; - if (lower.startsWith("gemini")) return 65536; - return 16384; +/** + * Documented output ceiling for a Google-surface model, or `undefined` when the id is not + * recognized. + * + * Unknown ids return `undefined` deliberately. An earlier revision returned a 16,384 floor for + * anything unmatched, which silently truncated aliases, gateway ids, and any model added after + * this table was written — the operator asked for N tokens and got 16,384 with no signal. A cap + * we cannot justify is worse than no cap: `structure/02_config-and-codex-home.md` is explicit + * that an explicit request value wins, so an unrecognized model passes through untouched and the + * upstream remains the authority on its own limit. + * + * Matching is prefix/family based rather than substring based for the same reason: `includes("pro")` + * matched any id containing "pro" (`my-prototype-model`), and `includes("oss")` matched any id + * containing "oss" (`crossover-v2`). + */ +export function maxOutputTokensForGoogleModel(modelId: string): number | undefined { + const lower = modelId.toLowerCase().trim(); + if (lower.startsWith("gemini")) { + // Pro tops out one token below the flash/other Gemini ceiling; both are documented values. + return /(^|[-.])pro([-.]|$)/.test(lower) ? 65535 : 65536; + } + if (lower.startsWith("claude")) return 64000; + if (lower.startsWith("gpt-oss")) return 32768; + return undefined; } export function clampGoogleMaxOutputTokens( @@ -64,6 +80,8 @@ export function clampGoogleMaxOutputTokens( ): number | undefined { if (requestedTokens === undefined || requestedTokens <= 0) return undefined; const modelMax = maxOutputTokensForGoogleModel(modelId); + // Unknown model: honour the request as-is rather than inventing a ceiling for it. + if (modelMax === undefined) return requestedTokens; return Math.min(requestedTokens, modelMax); } diff --git a/tests/google-output-clamp.test.ts b/tests/google-output-clamp.test.ts index 9f8f09718ae..f49aa41a2c5 100644 --- a/tests/google-output-clamp.test.ts +++ b/tests/google-output-clamp.test.ts @@ -9,7 +9,22 @@ describe("google maxOutputTokens clamp", () => { expect(maxOutputTokensForGoogleModel("claude-3-7-sonnet")).toBe(64000); expect(maxOutputTokensForGoogleModel("claude-3-5-sonnet@20241022")).toBe(64000); expect(maxOutputTokensForGoogleModel("gpt-oss-120b")).toBe(32768); - expect(maxOutputTokensForGoogleModel("custom-unknown-model")).toBe(16384); + }); + + test("an unrecognized model has no invented ceiling", () => { + // A cap we cannot justify silently truncates the operator's explicit request. Aliases, + // gateway ids, and models newer than this table must pass through untouched. + expect(maxOutputTokensForGoogleModel("custom-unknown-model")).toBeUndefined(); + expect(clampGoogleMaxOutputTokens("custom-unknown-model", 128000)).toBe(128000); + expect(maxOutputTokensForGoogleModel("some-gateway/gemini-3-pro")).toBeUndefined(); + }); + + test("family matching does not fire on incidental substrings", () => { + // "includes(pro)" matched my-prototype-model; "includes(oss)" matched crossover-v2. + expect(maxOutputTokensForGoogleModel("my-prototype-model")).toBeUndefined(); + expect(maxOutputTokensForGoogleModel("crossover-v2")).toBeUndefined(); + expect(maxOutputTokensForGoogleModel("gemini-3-pro-preview")).toBe(65535); + expect(maxOutputTokensForGoogleModel("gemini-3.5-flash")).toBe(65536); }); test("downward clamps excessive requested tokens to model max", () => {