From 509806dd614c09a877d85b8df050e3065b34b6ea Mon Sep 17 00:00:00 2001 From: Abhishek Divekar Date: Sun, 28 Jun 2026 04:50:02 -0700 Subject: [PATCH] fix(command-code): omit max_tokens when client omits it; correct registry caps The executor always sent params.max_tokens to /alpha/generate, fabricating a value from the registry maxOutputTokens when the client sent none. For DeepSeek V4 (registry maxOutputTokens: 384000) this produced a request the endpoint rejects with 400 "Too big: expected number to be <=200000 at params.max_tokens", breaking DeepSeek V4 Pro/Flash on command-code entirely. Root cause: getModelMaxTokensCap (added in #4518) fed the model's advertised output capacity (384000) straight into the request as the cap. That capacity is distinct from the endpoint's hard per-request ceiling of 200000, and is also simply wrong in the registry (Command Code's gateway caps DeepSeek output at 131072 per /provider/v1/models). Fix (executor): max_tokens is optional on /alpha/generate. Only forward it when the client actually supplies one, clamped to the 200000 endpoint ceiling so an oversized client value degrades gracefully instead of 400ing. When the client omits it, omit the field so upstream applies the model's native default. This mirrors the provider-driven clamp convention in antigravity.ts and removes the registry dependency from the request path (getModelMaxTokensCap deleted). Fix (registry): correct maxOutputTokens to the real Command Code gateway values from /provider/v1/models: DeepSeek V4 384000->131072, Kimi 131072->65536, GLM-5/5.1 131072->32768, MiniMax M2.5/M2.7 131072->65536, Qwen 3.6 131072->32768. These feed the combo router's output-limit check and dashboard metadata. The separate direct-DeepSeek spec in modelSpecs.ts (384000) is left untouched, since the model truly supports 384K output on its native API. Tests: omit-when-absent (GLM + DeepSeek, the reported scenario), clamp oversized client value to 200000, and honor a smaller client value unchanged. Co-authored-by: Cursor --- .../providers/registry/command-code/index.ts | 20 ++++---- open-sse/executors/commandCode.ts | 48 +++++++++++-------- tests/unit/command-code-executor.test.ts | 36 ++++++++++---- 3 files changed, 65 insertions(+), 39 deletions(-) diff --git a/open-sse/config/providers/registry/command-code/index.ts b/open-sse/config/providers/registry/command-code/index.ts index b3a35787b8f..6bc96c2372e 100644 --- a/open-sse/config/providers/registry/command-code/index.ts +++ b/open-sse/config/providers/registry/command-code/index.ts @@ -74,70 +74,70 @@ export const command_codeProvider: RegistryEntry = { name: "DeepSeek V4 Pro (CC)", supportsReasoning: true, contextLength: 1000000, - maxOutputTokens: 384000, + maxOutputTokens: 131072, }, { id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash (CC)", supportsReasoning: true, contextLength: 1000000, - maxOutputTokens: 384000, + maxOutputTokens: 131072, }, { id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6 (CC)", supportsReasoning: true, contextLength: 262144, - maxOutputTokens: 131072, + maxOutputTokens: 65536, }, { id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5 (CC)", supportsReasoning: true, contextLength: 262144, - maxOutputTokens: 131072, + maxOutputTokens: 65536, }, { id: "zai-org/GLM-5.1", name: "GLM-5.1 (CC)", supportsReasoning: true, contextLength: 200000, - maxOutputTokens: 131072, + maxOutputTokens: 32768, }, { id: "zai-org/GLM-5", name: "GLM-5 (CC)", supportsReasoning: true, contextLength: 200000, - maxOutputTokens: 131072, + maxOutputTokens: 32768, }, { id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7 (CC)", supportsReasoning: true, contextLength: 1048576, - maxOutputTokens: 131072, + maxOutputTokens: 65536, }, { id: "MiniMaxAI/MiniMax-M2.5", name: "MiniMax M2.5 (CC)", supportsReasoning: true, contextLength: 1048576, - maxOutputTokens: 131072, + maxOutputTokens: 65536, }, { id: "Qwen/Qwen3.6-Max-Preview", name: "Qwen 3.6 Max (CC)", supportsReasoning: true, contextLength: 1000000, - maxOutputTokens: 131072, + maxOutputTokens: 32768, }, { id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus (CC)", supportsReasoning: true, contextLength: 1000000, - maxOutputTokens: 131072, + maxOutputTokens: 32768, }, ], }; diff --git a/open-sse/executors/commandCode.ts b/open-sse/executors/commandCode.ts index 76fd7af3b88..a18100780cd 100644 --- a/open-sse/executors/commandCode.ts +++ b/open-sse/executors/commandCode.ts @@ -6,6 +6,12 @@ import { BaseExecutor, mergeUpstreamExtraHeaders, type ExecuteInput } from "./ba type JsonRecord = Record; export const COMMAND_CODE_VERSION = process.env.COMMAND_CODE_VERSION?.trim() || "0.33.2"; +// Hard server-side ceiling enforced by Command Code's /alpha/generate endpoint: +// any request with params.max_tokens > 200_000 is rejected with a 400 +// "Too big: expected number to be <=200000 at params.max_tokens". We only use +// this to clamp a CLIENT-SUPPLIED max_tokens down to a value the endpoint will +// accept; we never fabricate this number for requests that omit the field (see +// clampMaxTokens / buildCommandCodeBody). const MAX_COMMAND_CODE_TOKENS = 200_000; const encoder = new TextEncoder(); @@ -140,23 +146,16 @@ function convertMessages(messages: unknown): { system: string; messages: unknown return { system: system.join("\n\n"), messages: out }; } -function clampMaxTokens(value: unknown, cap: number = MAX_COMMAND_CODE_TOKENS): number { - const numeric = numberValue(value) ?? cap; - return Math.max(1, Math.min(Math.floor(numeric), cap)); -} - -// Resolve the per-model max_tokens cap for a given CommandCode model id. -// Falls back to MAX_COMMAND_CODE_TOKENS when the model isn't registered or -// when the registry entry omits maxOutputTokens. Without this, GLM-5.x -// requests get capped at 200_000 and rejected with "限制数值范围[1,131072]". -function getModelMaxTokensCap(modelId: string): number { - const entry = REGISTRY["command-code"]; - if (!entry) return MAX_COMMAND_CODE_TOKENS; - const model = entry.models?.find((m: { id: string }) => m.id === modelId); - const registryCap = (model as { maxOutputTokens?: number } | undefined)?.maxOutputTokens; - return typeof registryCap === "number" && registryCap > 0 - ? registryCap - : MAX_COMMAND_CODE_TOKENS; +// Clamp a client-supplied max_tokens to the endpoint ceiling, mirroring the +// provider-driven clamp in antigravity.ts: we only intervene when the value is +// present AND would otherwise be rejected (> 200_000). A valid value is +// returned floored; anything absent or non-numeric returns undefined so the +// caller can OMIT the field entirely and let Command Code's upstream apply the +// model's own native default (rather than us inventing a number). +function clampMaxTokens(value: unknown): number | undefined { + const numeric = numberValue(value); + if (numeric === undefined) return undefined; + return Math.max(1, Math.min(Math.floor(numeric), MAX_COMMAND_CODE_TOKENS)); } // Reasoning/thinking fields that payload rules or clients may inject and that @@ -188,13 +187,20 @@ function buildCommandCodeBody(model: string, body: unknown, stream = false): Jso messages: converted.messages, tools: convertTools(input.tools), system, - max_tokens: clampMaxTokens( - input.max_tokens ?? input.max_completion_tokens, - getModelMaxTokensCap(resolvedModel) - ), stream: true, }; + // Only forward max_tokens when the client actually supplied one. Omitting it + // lets Command Code's upstream apply the model's own native default, so we + // never invent a value (the old behavior, which sent the wrong number and got + // DeepSeek V4 rejected with "Too big: expected number to be <=200000"). When + // present, it is clamped to the endpoint ceiling so an oversized client value + // degrades gracefully instead of 400ing. + const maxTokens = clampMaxTokens(input.max_tokens ?? input.max_completion_tokens); + if (maxTokens !== undefined) { + params.max_tokens = maxTokens; + } + for (const field of COMMAND_CODE_PASSTHROUGH_FIELDS) { const value = input[field]; if (value !== undefined && value !== null) { diff --git a/tests/unit/command-code-executor.test.ts b/tests/unit/command-code-executor.test.ts index a52f59b62b4..35ad2632574 100644 --- a/tests/unit/command-code-executor.test.ts +++ b/tests/unit/command-code-executor.test.ts @@ -294,40 +294,60 @@ test("Command Code executor surfaces upstream and streamed errors", async () => }, /boom/); }); -test("Command Code executor caps max_tokens to the registered per-model limit (GLM-5.x)", async () => { +test("Command Code executor omits max_tokens when the client does not supply one (GLM-5.x)", async () => { const calls: FetchCall[] = []; globalThis.fetch = async (url, init = {}) => { calls.push({ url: String(url), init, body: JSON.parse(String(init.body)) }); return commandCodeStream([{ type: "text-delta", text: "ok" }, { type: "finish" }]); }; - // GLM-5 and GLM-5.1 are registered with maxOutputTokens: 131072. - // Without per-model capping, the upstream rejects with - // "限制数值范围[1,131072]". + // No client max_tokens: we must NOT fabricate one. Omitting the field lets + // Command Code's upstream apply the model's own native default. await getExecutor("command-code").execute({ model: "zai-org/GLM-5.1", stream: false, credentials: { apiKey: "cc_test_key" }, body: { messages: [{ role: "user", content: "Hi" }] }, }); - assert.equal(calls[0].body.params.max_tokens, 131072); + assert.ok(!("max_tokens" in calls[0].body.params)); }); -test("Command Code executor caps max_tokens to the registered per-model limit (DeepSeek v4)", async () => { +test("Command Code executor omits max_tokens for DeepSeek v4 when the client does not supply one", async () => { const calls: FetchCall[] = []; globalThis.fetch = async (url, init = {}) => { calls.push({ url: String(url), init, body: JSON.parse(String(init.body)) }); return commandCodeStream([{ type: "text-delta", text: "ok" }, { type: "finish" }]); }; - // DeepSeek v4 pro is registered with maxOutputTokens: 384000. + // Regression: previously the executor invented max_tokens from the registry + // (384000), which /alpha/generate rejects with a 400 + // "Too big: expected number to be <=200000". With no client value we now omit + // the field entirely, so the request succeeds and upstream picks the default. await getExecutor("command-code").execute({ model: "deepseek/deepseek-v4-pro", stream: false, credentials: { apiKey: "cc_test_key" }, body: { messages: [{ role: "user", content: "Hi" }] }, }); - assert.equal(calls[0].body.params.max_tokens, 384000); + assert.ok(!("max_tokens" in calls[0].body.params)); +}); + +test("Command Code executor clamps an oversized client-supplied max_tokens to the endpoint ceiling", async () => { + const calls: FetchCall[] = []; + globalThis.fetch = async (url, init = {}) => { + calls.push({ url: String(url), init, body: JSON.parse(String(init.body)) }); + return commandCodeStream([{ type: "text-delta", text: "ok" }, { type: "finish" }]); + }; + + // A client asking for more than the 200000 endpoint ceiling is clamped down + // (not 400'd), mirroring the provider-driven clamp in antigravity.ts. + await getExecutor("command-code").execute({ + model: "deepseek/deepseek-v4-pro", + stream: false, + credentials: { apiKey: "cc_test_key" }, + body: { messages: [{ role: "user", content: "Hi" }], max_tokens: 500000 }, + }); + assert.equal(calls[0].body.params.max_tokens, 200000); }); test("Command Code executor honors a smaller client-provided max_tokens under the per-model cap", async () => {