diff --git a/src/integrations/runtimeMetadata.test.ts b/src/integrations/runtimeMetadata.test.ts index a871bba8ef..042da33dd2 100644 --- a/src/integrations/runtimeMetadata.test.ts +++ b/src/integrations/runtimeMetadata.test.ts @@ -199,7 +199,29 @@ describe('resolveOpenAIShimRuntimeContext - GLM on a non-Z.AI gateway (#1896)', expect(result.openaiShimConfig.preserveReasoningContent).toBe(true) // tool_stream is Z.AI-proprietary and must NOT be inferred; NVIDIA NIM (and // other third-party gateways) reject it with 400 Unsupported parameter(s). - expect(result.openaiShimConfig.enableToolStreaming).not.toBe(true) + expect(result.openaiShimConfig.enableToolStreaming).toBe(false) + }) +}) + +describe('resolveOpenAIShimRuntimeContext - NVIDIA NIM GLM-5.2 (regression #1950)', () => { + // The user selected `z-ai/glm-5.2` from NVIDIA NIM's discovered (dynamic) + // model catalog. Even when a GLM catalog entry exists on a non-Z.AI gateway, + // `tool_stream` must stay off (Z.AI-proprietary); the reasoning-shaping shim + // still applies because GLM needs it on any gateway. + it('does not enable tool_stream for NVIDIA NIM GLM-5.2 and keeps the reasoning shim', () => { + const result = resolveOpenAIShimRuntimeContext({ + model: 'z-ai/glm-5.2', + baseUrl: 'https://integrate.api.nvidia.com/v1', + processEnv: { NVIDIA_NIM: '1' }, + }) + + expect(result.routeId).toBe('nvidia-nim') + expect(result.openaiShimConfig.enableToolStreaming).toBe(false) + expect(result.openaiShimConfig.thinkingRequestFormat).toBe('zai-compatible') + expect(result.openaiShimConfig.preserveReasoningContent).toBe(true) + expect(result.openaiShimConfig.requireReasoningContentOnAssistantMessages).toBe(true) + expect(result.openaiShimConfig.maxTokensField).toBe('max_tokens') + expect(result.openaiShimConfig.removeBodyFields).toContain('store') }) }) diff --git a/src/services/api/claude.streamWatchdog.test.ts b/src/services/api/claude.streamWatchdog.test.ts index 7903ab9022..3ccbbbb38d 100644 --- a/src/services/api/claude.streamWatchdog.test.ts +++ b/src/services/api/claude.streamWatchdog.test.ts @@ -1,10 +1,9 @@ import { - afterAll, afterEach, beforeEach, describe, expect, - mock, + spyOn, test, } from 'bun:test' import type { @@ -51,22 +50,25 @@ type CreateHandler = (...args: CreateArgs) => unknown let createHandler: CreateHandler | undefined let importCounter = 0 - -mock.module('./client.js', () => ({ - CLIENT_REQUEST_ID_HEADER: actualClientModule.CLIENT_REQUEST_ID_HEADER, - getAnthropicClient: async () => ({ - beta: { - messages: { - create: (...args: CreateArgs) => { - if (!createHandler) { - throw new Error('test client create handler not configured') - } - return createHandler(...args) +let restoreClientSpy: (() => void) | undefined + +function installClientSpy(): void { + const clientSpy = spyOn(actualClientModule, 'getAnthropicClient').mockImplementation( + async () => ({ + beta: { + messages: { + create: (...args: CreateArgs) => { + if (!createHandler) { + throw new Error('test client create handler not configured') + } + return createHandler(...args) + }, }, }, - }, - }), -})) + }) as never, + ) + restoreClientSpy = () => clientSpy.mockRestore() +} function makeBetaMessage( id: string, @@ -291,6 +293,7 @@ function setTestMacro(): void { beforeEach(async () => { await acquireSharedMutationLock('claude.streamWatchdog.test.ts') + installClientSpy() setTestMacro() for (const key of envKeys) { delete process.env[key] @@ -305,6 +308,8 @@ beforeEach(async () => { afterEach(() => { try { + restoreClientSpy?.() + restoreClientSpy = undefined createHandler = undefined for (const key of envKeys) { const envKey: string = key @@ -336,10 +341,6 @@ afterEach(() => { } }) -afterAll(() => { - mock.module('./client.js', () => actualClientModule) -}) - describe('Claude stream watchdog', () => { test('falls back when the top-level stream iterator never settles', async () => { const wedged = makeWedgedStream() diff --git a/src/services/api/errors.openaiCompatibility.test.ts b/src/services/api/errors.openaiCompatibility.test.ts index 7d037fe30d..db366c640f 100644 --- a/src/services/api/errors.openaiCompatibility.test.ts +++ b/src/services/api/errors.openaiCompatibility.test.ts @@ -128,3 +128,21 @@ test('maps tool_call_incompatible category markers to model/tool guidance', () = expect(text).toContain('rejected tool-calling payloads') expect(text).toContain('/model') }) + +test('maps tool_stream_unsupported without promising a retry after failure', () => { + const error = APIError.generate( + 400, + undefined, + 'OpenAI API error 400: tool_stream is unsupported [openai_category=tool_stream_unsupported]', + new Headers(), + ) + + const message = getAssistantMessageFromError(error, 'glm-5.2') + const text = getFirstText(message) + + expect(text).toContain('rejected the `tool_stream` parameter') + expect(text).toContain('cannot be streamed') + expect(text).toContain('switch models') + expect(text).toMatch(/(\/model|--model)/) + expect(text).not.toContain('Retrying') +}) diff --git a/src/services/api/errors.ts b/src/services/api/errors.ts index 36a11d16d4..ae2ab7eb35 100644 --- a/src/services/api/errors.ts +++ b/src/services/api/errors.ts @@ -148,6 +148,12 @@ function mapOpenAICompatibilityFailureToAssistantMessage(options: { error: 'invalid_request', }) + case 'tool_stream_unsupported': + return createAssistantAPIErrorMessage({ + content: `The selected provider rejected the \`tool_stream\` parameter. Tool calls cannot be streamed on this provider. If this persists, switch models via ${switchCmd}.`, + error: 'invalid_request', + }) + case 'malformed_provider_response': return createAssistantAPIErrorMessage({ content: `${API_ERROR_MESSAGE_PREFIX}: Provider returned a malformed response. Confirm endpoint compatibility and check local proxy/network middleware.`, diff --git a/src/services/api/openaiErrorClassification.test.ts b/src/services/api/openaiErrorClassification.test.ts index 662f4e31f9..0d6c13821f 100644 --- a/src/services/api/openaiErrorClassification.test.ts +++ b/src/services/api/openaiErrorClassification.test.ts @@ -148,6 +148,232 @@ test('classifies tool compatibility failures', () => { expect(failure.category).toBe('tool_call_incompatible') }) +test('classifies tool_stream rejection as tool_stream_unsupported (#1950)', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Validation: Unsupported parameter(s): `tool_stream`', + }) + + expect(failure.category).toBe('tool_stream_unsupported') + expect(failure.retryable).toBe(false) +}) + +test('prioritizes tool_stream rejection over accompanying tool-call wording', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Invalid parameter tool_stream; tool_calls are not supported by this model', + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('classifies OpenAI-style unrecognized tool_stream arguments', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Unrecognized request argument supplied: tool_stream', + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('classifies a tool_stream parameter rejection that is conditional on function calls', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Invalid parameter tool_stream in function calls', + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('classifies a top-level tool_stream rejection that explains function parameters', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Unsupported parameter tool_stream in function parameters', + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test.each([ + "Unknown parameter: 'tool_stream'", + 'Invalid parameter: "tool_stream"', + "Parameter 'tool_stream' is not supported", + 'Unsupported parameters: tool_stream', + 'Unsupported parameter(s): ["tool_stream"]', + 'Unsupported parameter: (tool_stream)', + 'Unknown parameters: [tool_stream]', + "Parameter 'tool_stream' is unknown", + "'tool_stream' is an unknown parameter", + 'Invalid "tool_stream" parameter', + 'tool_stream is unsupported', + 'Unsupported parameter(s): tool_stream. Tools are available only in non-streaming mode.', + '{"error":{"message":"tool_stream is unsupported"}}', + '{"error":{"message":"tool_stream is not supported"}}', + '{"error":{"message":"Unknown parameter","param":"tool_stream"}}', + '{"error":{"message":"Invalid parameter","param":"tool_stream"}}', + '{"detail":[{"type":"extra_forbidden","loc":["body","tool_stream"],"msg":"Extra inputs are not permitted","input":true}]}', + '{"detail":[{"type":"value_error.extra","loc":["body","tool_stream"],"msg":"extra fields not permitted"}]}', + 'Additional properties are not allowed ("tool_stream" was unexpected)', +])('classifies quoted tool_stream parameter rejections: %s', body => { + const failure = classifyOpenAIHttpFailure({ status: 400, body }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('classifies a FastAPI validation rejection at its normal 422 status', () => { + const failure = classifyOpenAIHttpFailure({ + status: 422, + body: '{"detail":[{"type":"extra_forbidden","loc":["body","tool_stream"],"msg":"Extra inputs are not permitted","input":true}]}', + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('classifies a root structured tool_stream unsupported message', () => { + const failure = classifyOpenAIHttpFailure({ + status: 422, + body: JSON.stringify({ + detail: [{ + type: 'value_error', + loc: ['body', 'tool_stream'], + msg: 'tool_stream is unsupported', + }], + }), + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('classifies a root tool_stream extra-field rejection alongside tool validation details', () => { + const failure = classifyOpenAIHttpFailure({ + status: 422, + body: JSON.stringify({ + detail: [ + { + type: 'extra_forbidden', + loc: ['body', 'tool_stream'], + msg: 'Extra inputs are not permitted', + }, + { + type: 'missing', + loc: ['body', 'tools'], + msg: 'Field required', + }, + ], + }), + }) + + expect(failure.category).toBe('tool_stream_unsupported') +}) + +test('does not classify a generic 400 as tool_stream_unsupported', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Invalid request: missing required field `messages`', + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test.each([ + 'Tool "tool_stream" is unsupported', + "Function 'tool_stream' is invalid", + 'Tool: tool_stream is unsupported', + 'Function: tool_stream is invalid', + 'tool_stream is unsupported as a function', + 'tool_stream is unsupported as a tool', + 'Additional properties are not allowed in function tool_stream', + 'The tool named "tool_stream" is unsupported', + 'Function name tool_stream is invalid', +])('does not classify a tool name error as a tool_stream parameter rejection: %s', body => { + const failure = classifyOpenAIHttpFailure({ status: 400, body }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test('does not classify an invalid schema for a tool named tool_stream as a parameter rejection', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: "Invalid schema for function 'tool_stream': properties must be an object", + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test('does not classify a raw tool-schema property error as a parameter rejection', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: "Invalid schema for function 'Bash': Additional properties are not allowed ('tool_stream' was unexpected)", + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test('does not classify a tool-schema error whose location follows the parameter name', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: 'Additional properties are not allowed (tool_stream was unexpected) at body.tools.0.function.parameters.properties', + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test.each([ + 'Invalid schema: param=tool_stream', + 'Malformed tool schema: unexpected property tool_stream', + 'Additional properties are not allowed: tool_stream in tool definition', + 'Invalid parameter tool_stream in function definition', + 'Additional properties are not allowed (tool_stream was unexpected) in the function parameters', + 'Invalid parameter tool_stream in function Bash', + 'Invalid parameter tool_stream in the function Bash', + 'Invalid parameter tool_stream for tool Bash', + 'At body.tools[0].function.parameters: Extra inputs are not permitted: tool_stream', + 'Unexpected field tool_stream in tool schema', + 'Extra inputs are not permitted: tool_stream in function parameters', + 'tool_stream unexpected field in tool schema', +])('does not classify a generic schema diagnostic as a parameter rejection: %s', body => { + const failure = classifyOpenAIHttpFailure({ status: 400, body }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test('does not classify a structured validation error that merely references tool_stream', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: '{"error":{"message":"Parameter is required","param":"tool_stream"}}', + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test.each([ + '{"detail":[{"type":"missing","loc":["body","tool_stream"],"msg":"Field required"}]}', + '{"detail":[{"type":"string_type","loc":["body","tool_stream"],"msg":"Input should be a valid string"}]}', + '{"detail":[{"type":"extra_forbidden","loc":["body","tool_stream","mode"],"msg":"Extra inputs are not permitted"}]}', +])('does not classify a structured validation error for a supported tool_stream field: %s', body => { + const failure = classifyOpenAIHttpFailure({ status: 400, body }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test('does not classify a structured validation error for a tool named tool_stream', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: '{"detail":[{"type":"extra_forbidden","loc":["body","tools",0,"function","name"],"msg":"Extra inputs are not permitted","input":"tool_stream"}]}', + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + +test('does not classify a structured validation error for a tool-schema property named tool_stream', () => { + const failure = classifyOpenAIHttpFailure({ + status: 400, + body: '{"detail":[{"type":"extra_forbidden","loc":["body","tools",0,"function","parameters","properties","tool_stream"],"msg":"Extra inputs are not permitted","input":{}}]}', + }) + + expect(failure.category).not.toBe('tool_stream_unsupported') +}) + test('embeds and extracts category markers in formatted messages', () => { const marker = formatOpenAICategoryMarker('endpoint_not_found') expect(marker).toBe('[openai_category=endpoint_not_found]') diff --git a/src/services/api/openaiErrorClassification.ts b/src/services/api/openaiErrorClassification.ts index 3413a6a57d..dff43ccc25 100644 --- a/src/services/api/openaiErrorClassification.ts +++ b/src/services/api/openaiErrorClassification.ts @@ -11,6 +11,7 @@ export type OpenAICompatibilityFailureCategory = | 'vision_not_supported' | 'context_overflow' | 'tool_call_incompatible' + | 'tool_stream_unsupported' | 'malformed_provider_response' | 'provider_unavailable' | 'unknown' @@ -44,6 +45,7 @@ const OPENAI_COMPATIBILITY_FAILURE_CATEGORIES: ReadonlySet = [] + for (const detail of parsed.detail) { + if (!detail || typeof detail !== 'object') continue + const { loc, type, msg } = detail as { loc?: unknown; type?: unknown; msg?: unknown } + if (Array.isArray(loc)) details.push({ loc, type, msg }) + } + const isRootToolStreamLocation = (loc: unknown[]): boolean => + loc[0] === 'body' && loc[1] === 'tool_stream' + const isRootToolStreamExtraField = (detail: { + loc: unknown[] + type?: unknown + msg?: unknown + }): boolean => + detail.loc.length === 2 && + isRootToolStreamLocation(detail.loc) && + ( + detail.type === 'extra_forbidden' || + detail.type === 'value_error.extra' || + (typeof detail.msg === 'string' && /extra (?:inputs|fields) (?:are )?not permitted/i.test(detail.msg)) + ) + if (details.some(isRootToolStreamExtraField)) return true + if ( + details.some( + detail => + detail.loc.length === 2 && + isRootToolStreamLocation(detail.loc) && + typeof detail.msg === 'string' && + /tool_stream is (?:unsupported|not supported|unknown|invalid)/i.test(detail.msg), + ) + ) return true + if ( + details.some(detail => + detail.loc.includes('tools') || isRootToolStreamLocation(detail.loc) + ) + ) return false + return undefined + } catch { + return undefined + } +} + +// Detect a gateway rejecting the Z.AI-proprietary `tool_stream` parameter +// (e.g. NVIDIA NIM: `400 Unsupported parameter(s): tool_stream`). The +// `tool_call` substring in isToolCompatibilityMessage does NOT match +// `tool_stream`, so this needs its own matcher. +function isToolStreamUnsupportedMessage(body: string): boolean { + const normalized = body.toLowerCase().replace(/['"`]/g, '') + const structuredValidation = getStructuredToolStreamValidationError(body) + if ( + /(?:function|tool)\s*:?\s+tool_stream\b/.test(normalized) || + /\b(?:function|tool)\b.*?\b(?:schema|properties?)\b.*?\btool_stream\b/.test(normalized) || + /\b(?:invalid|malformed)\s+(?:tool\s+)?schema\b.*?\btool_stream\b/.test(normalized) || + /\btool_stream\b.*?\b(?:invalid|malformed)\s+(?:tool\s+)?schema\b/.test(normalized) || + /\b(?:tool|function)\s+definition\b.*?\btool_stream\b/.test(normalized) || + /\btool_stream\b.*?\b(?:tool|function)\s+definition\b/.test(normalized) || + /\btool_stream\b.*?\b(?:body\.)?tools?\s*(?:\.|\[)/.test(normalized) || + /\b(?:body\.)?tools?\s*(?:\.|\[).*?\btool_stream\b/.test(normalized) || + /(?:unexpected (?:field|property|parameter)|extra[_\s-]?forbidden|extra inputs are not permitted|additional properties? (?:are )?not allowed).*?\btool_stream\b.*?\b(?:in|at|for)\s+(?:(?:an?|the)\s+)?(?:tool|function)?\s*(?:schema|parameters?|properties?)\b/.test(normalized) || + /\btool_stream\b.*?(?:unexpected (?:field|property|parameter)|extra[_\s-]?forbidden|extra inputs are not permitted|additional properties? (?:are )?not allowed).*?\b(?:tool|function)\s+(?:schema|parameters?|properties?)\b/.test(normalized) || + /\b(?:invalid|malformed)\s+parameter\s+tool_stream\b.*?\b(?:in|for)\s+(?:(?:an?|the)\s+)?(?:function|tool)\s+(?!calls?\b|calling\b)\S+/.test(normalized) || + /\badditional properties?\b.*?\btool_stream\b.*?\b(?:in|for)\s+(?:(?:an?|the)\s+)?(?:function|tool)\s+(?!calls?\b|calling\b)\S+/.test(normalized) || + structuredValidation === false + ) return false + if (structuredValidation === true) return true + return ( + /(?:unsupported|unknown|unrecognized|invalid)\s+(?:request\s+argument(?:\s+supplied)?|parameter(?:s|\(s\))?)(?:\s*[:=])?\s*(?:[\[(<]\s*)?tool_stream\b(?:\s*[\])>])?/.test(normalized) || + /(?:request\s+argument(?:\s+supplied)?|parameter(?:s|\(s\))?)\s+(?:[\[(<]\s*)?tool_stream\b(?:\s*[\])>])?\s+(?:is\s+)?(?:unsupported|not\s+supported|unknown|invalid)\b/.test(normalized) || + /tool_stream\s+(?:is\s+)?(?:an?\s+)?(?:unsupported|not\s+supported|unknown|invalid)\s+(?:request\s+argument|parameter(?:s|\(s\))?)\b/.test(normalized) || + /(?:unsupported|unknown|unrecognized|invalid)\s+tool_stream\s+(?:request\s+argument|parameter(?:s|\(s\))?)\b/.test(normalized) || + /(?:^|\n|\bmessage\s*:\s*)\s*tool_stream\s+(?:is\s+)?(?:an?\s+)?(?:unsupported|not\s+supported|unknown|invalid)\b(?!\s+as\s+(?:a\s+)?(?:function|tool)\b)/.test(normalized) || + /(?:unsupported|unknown|unrecognized|invalid|not\s+supported).*?\bparam(?:eter)?\s*[:=]\s*tool_stream\b/.test(normalized) || + /\bparam(?:eter)?\s*[:=]\s*tool_stream\b.*?(?:unsupported|unknown|unrecognized|invalid|not\s+supported)/.test(normalized) || + /(?:extra[_\s-]?forbidden|extra inputs are not permitted|additional properties? (?:are )?not allowed|unexpected (?:field|property|parameter)).*?tool_stream\b/.test(normalized) || + /tool_stream\b.*?(?:extra[_\s-]?forbidden|extra inputs are not permitted|unexpected (?:field|property|parameter))/.test(normalized) + ) +} + function isMalformedProviderResponse(body: string): boolean { const lower = body.toLowerCase() return ( @@ -458,6 +541,27 @@ export function classifyOpenAIHttpFailure(options: { } } + // `tool_stream` is a Z.AI-proprietary streaming extension. Some OpenAI- + // compatible gateways (e.g. NVIDIA NIM) reject it with a 400 like + // "Unsupported parameter(s): `tool_stream`". Classify it distinctly so the + // shim can self-heal by dropping just `tool_stream` and retrying with tools + // intact (issue #1950). Match liberally on the parameter name plus an + // unsupported/unknown-parameter signal so provider-specific wording still + // triggers the fallback. + if ( + (options.status === 400 || options.status === 422) && + isToolStreamUnsupportedMessage(body) + ) { + return { + source: 'http', + category: 'tool_stream_unsupported', + retryable: false, + status: options.status, + message: body, + hint: 'Provider rejected the `tool_stream` parameter. Retrying without it (tool calls are not streamed).', + } + } + if (options.status === 400 && isToolCompatibilityMessage(body)) { return { source: 'http', diff --git a/src/services/api/openaiShim.test.ts b/src/services/api/openaiShim.test.ts index 067cf44177..a04c8fb14c 100644 --- a/src/services/api/openaiShim.test.ts +++ b/src/services/api/openaiShim.test.ts @@ -4111,6 +4111,37 @@ test('OPENAI_API_KEYS rotates to the next key on rate-limit failure', async () = expect(authorizations).toEqual(['Bearer key-a', 'Bearer key-b']) }) +test('OPENAI_API_KEYS does not reuse a cooled-down key after every key is rate-limited', async () => { + const authorizations: Array = [] + + process.env.CLAUDE_CODE_USE_OPENAI = '1' + process.env.OPENAI_BASE_URL = 'https://api.openai.com/v1' + process.env.OPENAI_MODEL = 'gpt-5.5' + process.env.OPENAI_API_KEYS = 'key-a,key-b' + delete process.env.OPENAI_API_KEY + + globalThis.fetch = (async (_input, init) => { + const headers = init?.headers as Record | undefined + authorizations.push(headers?.Authorization ?? headers?.authorization ?? null) + return new Response(JSON.stringify({ error: { message: 'rate limited' } }), { + status: 429, + headers: { 'Content-Type': 'application/json' }, + }) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + await expect( + client.beta.messages.create({ + model: 'gpt-5.5', + messages: [{ role: 'user', content: 'hello' }], + max_tokens: 32, + stream: false, + }), + ).rejects.toThrow() + + expect(authorizations).toEqual(['Bearer key-a', 'Bearer key-b']) +}) + test('comma-separated OPENAI_API_KEY rotates to the next key on rate-limit failure', async () => { const authorizations: Array = [] @@ -8743,6 +8774,210 @@ test('NVIDIA NIM Z.AI GLM omits chat template thinking kwargs when thinking is d expect(requestBody?.chat_template_kwargs).toBeUndefined() }) +// Regression test for #1950: GLM-5.2 served through NVIDIA NIM +// (`integrate.api.nvidia.com`) must never receive the Z.AI-proprietary +// `tool_stream` parameter. Streaming tool calls are simply not streamed on +// this gateway; sending the parameter aborts the request with +// `400 Unsupported parameter(s): tool_stream`. +test('NVIDIA NIM Z.AI GLM streaming request with tools does not send tool_stream (regression #1950)', async () => { + process.env.OPENAI_BASE_URL = 'https://integrate.api.nvidia.com/v1' + process.env.NVIDIA_API_KEY = 'nvapi-test' + + let requestBody: Record | undefined + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + return makeSseResponse(makeStreamChunks([ + { + id: 'chatcmpl-1', + object: 'chat.completion.chunk', + model: 'z-ai/glm-5.2', + choices: [{ index: 0, delta: { content: 'ok' }, finish_reason: null }], + }, + { + id: 'chatcmpl-1', + object: 'chat.completion.chunk', + model: 'z-ai/glm-5.2', + choices: [{ index: 0, delta: {}, finish_reason: 'stop' }], + }, + ])) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + await client.beta.messages.create({ + model: 'z-ai/glm-5.2', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [ + { + name: 'Bash', + description: 'Run a shell command', + input_schema: { + type: 'object', + properties: { command: { type: 'string' } }, + required: ['command'], + }, + }, + ], + max_tokens: 64, + stream: true, + }) + + // tool_stream is a Z.AI-only streaming extension; NVIDIA NIM rejects it with + // `400 Unsupported parameter(s): tool_stream`. Streaming tool calls simply + // aren't streamed on this gateway. + expect(requestBody?.tool_stream).toBeUndefined() +}) + +// Regression test for #1950: even if a gateway rejects `tool_stream` with a +// 400 (e.g. NVIDIA NIM: `Unsupported parameter(s): tool_stream`), the shim +// self-heals by dropping only that parameter and retrying with tools intact. +// Here we exercise the generic self-heal using a Z.AI-contract gateway that +// actually sends `tool_stream`, then rejects it — proving the retry drops the +// parameter rather than surfacing a hard error. +test('Shim self-heals a JSON `tool_stream` rejection by retrying without it (#1950)', async () => { + process.env.OPENAI_BASE_URL = 'https://api.z.ai/api/coding/paas/v4' + process.env.OPENAI_API_KEY = 'sk-zai-test' + + const requestBodies: Array> = [] + let callCount = 0 + globalThis.fetch = (async (_input, init) => { + requestBodies.push(JSON.parse(String(init?.body))) + callCount += 1 + if (callCount === 1) { + return new Response( + '{"error":{"message":"tool_stream is unsupported"}}', + { status: 400, headers: { 'Content-Type': 'application/json' } }, + ) + } + return makeSseResponse(makeStreamChunks([ + { + id: 'chatcmpl-1', + object: 'chat.completion.chunk', + model: 'glm-5.2', + choices: [{ index: 0, delta: { content: 'ok' }, finish_reason: null }], + }, + { + id: 'chatcmpl-1', + object: 'chat.completion.chunk', + model: 'glm-5.2', + choices: [{ index: 0, delta: {}, finish_reason: 'stop' }], + }, + ])) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + // Must not throw — the self-heal retry succeeds. + await client.beta.messages.create({ + model: 'glm-5.2', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [ + { + name: 'Bash', + description: 'Run a shell command', + input_schema: { + type: 'object', + properties: { command: { type: 'string' } }, + required: ['command'], + }, + }, + ], + max_tokens: 64, + stream: true, + }) + + // First attempt sent tool_stream; the self-heal dropped it and retried. + expect(requestBodies).toHaveLength(2) + expect(requestBodies[0]?.tool_stream).toBe(true) + expect(requestBodies[1]?.tool_stream).toBeUndefined() + // Tools are preserved across the retry. + expect(Array.isArray(requestBodies[1]?.tools)).toBe(true) +}) + +test('Shim stops after one tool_stream self-heal retry when the retry also fails (#1950)', async () => { + process.env.OPENAI_BASE_URL = 'https://api.z.ai/api/coding/paas/v4' + process.env.OPENAI_API_KEY = 'sk-zai-test' + + const requestBodies: Array> = [] + globalThis.fetch = (async (_input, init) => { + requestBodies.push(JSON.parse(String(init?.body))) + return new Response( + '{"error":{"message":"tool_stream is unsupported"}}', + { status: 400, headers: { 'Content-Type': 'application/json' } }, + ) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + await expect( + client.beta.messages.create({ + model: 'glm-5.2', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [{ + name: 'Bash', + description: 'Run a shell command', + input_schema: { + type: 'object', + properties: { command: { type: 'string' } }, + required: ['command'], + }, + }], + max_tokens: 64, + stream: true, + }), + ).rejects.toThrow() + + expect(requestBodies).toHaveLength(2) + expect(requestBodies[0]?.tool_stream).toBe(true) + expect(requestBodies[1]?.tool_stream).toBeUndefined() +}) + +test('Shim retries a tool_stream rejection with the same pooled credential (#1950)', async () => { + process.env.OPENAI_BASE_URL = 'https://api.z.ai/api/coding/paas/v4' + process.env.OPENAI_API_KEYS = 'key-a,key-b' + delete process.env.OPENAI_API_KEY + + const authorizations: Array = [] + let callCount = 0 + globalThis.fetch = (async (_input, init) => { + const headers = init?.headers as Record | undefined + authorizations.push(headers?.Authorization ?? headers?.authorization ?? null) + callCount += 1 + if (callCount === 1) { + return new Response( + '{"error":{"message":"Validation: Unsupported parameter(s): `tool_stream`"}}', + { status: 400, headers: { 'Content-Type': 'application/json' } }, + ) + } + return makeSseResponse(makeStreamChunks([ + { + id: 'chatcmpl-1', + object: 'chat.completion.chunk', + model: 'glm-5.2', + choices: [{ index: 0, delta: { content: 'ok' }, finish_reason: null }], + }, + { + id: 'chatcmpl-1', + object: 'chat.completion.chunk', + model: 'glm-5.2', + choices: [{ index: 0, delta: {}, finish_reason: 'stop' }], + }, + ])) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + await client.beta.messages.create({ + model: 'glm-5.2', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [{ + name: 'Bash', + description: 'Run a shell command', + input_schema: { type: 'object', properties: { command: { type: 'string' } }, required: ['command'] }, + }], + max_tokens: 64, + stream: true, + }) + + expect(authorizations).toEqual(['Bearer key-a', 'Bearer key-a']) +}) + test('Z.AI GLM-5.2: streaming requests with tools send tool_stream', async () => { process.env.OPENAI_BASE_URL = 'https://api.z.ai/api/coding/paas/v4' process.env.OPENAI_API_KEY = 'sk-zai-test' diff --git a/src/services/api/openaiShim.ts b/src/services/api/openaiShim.ts index 40b2f15710..0f7aa9b30a 100644 --- a/src/services/api/openaiShim.ts +++ b/src/services/api/openaiShim.ts @@ -4488,6 +4488,8 @@ class OpenAIShimMessages { let requestUrl = buildRequestUrl(activeBaseUrl) const attemptedLocalBaseUrls = new Set([activeBaseUrl]) let didRetryWithoutTools = false + let didRetryWithoutToolStream = false + let retryCredentialLease: CredentialLease | null = null let didRefreshCopilotToken = false let refreshedCopilotToken: string | undefined @@ -4592,7 +4594,7 @@ class OpenAIShimMessages { ? localRetryBaseUrls.length + 1 : 0 const credentialPoolAttempts = credentialPool?.size ?? 1 - const maxAttempts = + let maxAttempts = Math.max(isGithub ? GITHUB_429_MAX_RETRIES : 1, credentialPoolAttempts) + maxSelfHealAttempts @@ -4678,7 +4680,8 @@ class OpenAIShimMessages { : 'openai' const { correlationId, startTime } = logApiCallStart(provider, request.resolvedModel) for (let attempt = 0; attempt < maxAttempts; attempt++) { - const credentialLease = credentialPool?.next() ?? null + const credentialLease = retryCredentialLease ?? credentialPool?.next() ?? null + retryCredentialLease = null if (credentialPool && !credentialLease) { throw APIError.generate( 401, @@ -4945,6 +4948,36 @@ class OpenAIShimMessages { continue } + // `tool_stream` self-heal (#1950): some OpenAI-compatible gateways (e.g. + // NVIDIA NIM) reject the Z.AI-proprietary `tool_stream` parameter with a + // 400. Drop only that parameter and retry with tools intact — streaming + // tool calls simply aren't streamed on such gateways. This guards against + // regressions where the parameter slips through the catalog/runtime + // gating that normally suppresses it. + if ( + !didRetryWithoutToolStream && + failure.category === 'tool_stream_unsupported' && + body.tool_stream === true + ) { + didRetryWithoutToolStream = true + // Reserve one additional request only after this specific recovery is + // needed. Increasing the shared initial budget changes unrelated + // GitHub and credential-pool retry behavior. + maxAttempts += 1 + delete body.tool_stream + refreshSerializedBody() + // This retry only changes request formatting. Reuse the credential that + // received the rejection so a pool with unequal model access cannot + // turn a recoverable 400 into an unrelated authorization failure. + retryCredentialLease = credentialLease + + logForDebugging( + `[OpenAIShim] self-heal retry reason=tool_stream_unsupported method=POST url=${redactUrlForDiagnostics(requestUrl)} model=${request.resolvedModel}`, + { level: 'warn' }, + ) + continue + } + let errorResponse: object | undefined try { errorResponse = JSON.parse(errorBody) } catch { /* raw text */ } throwClassifiedHttpError(