diff --git a/.env.example b/.env.example index 5d5aad5970..90ebea1f72 100644 --- a/.env.example +++ b/.env.example @@ -304,6 +304,23 @@ ANTHROPIC_API_KEY=sk-ant-your-key-here # OPENAI_MODEL=your-model-id-here +# ----------------------------------------------------------------------------- +# Option 6b: llama-server (llama.cpp) / self-hosted OpenAI-compat +# ----------------------------------------------------------------------------- +# Prefer a provider profile (/provider) with "Self-hosted tools" enabled — +# no shell env required. The flag is stored on that profile only. +# +# Prefer structured tools on the server when possible, e.g.: +# llama-server -m model.gguf --jinja --port 8080 +# (use the tool-call parser appropriate for your model, e.g. Qwen) +# +# CLAUDE_CODE_USE_OPENAI=1 +# OPENAI_BASE_URL=http://127.0.0.1:8080/v1 +# OPENAI_API_KEY=none +# OPENAI_MODEL=qwen3.6:35b +# OPENAI_SELF_HOSTED_TOOLS=1 # optional shell override; prefer profile flag + + # ----------------------------------------------------------------------------- # Option 7: AWS Bedrock # ----------------------------------------------------------------------------- diff --git a/src/components/ProviderManager.test.tsx b/src/components/ProviderManager.test.tsx index 6d1fd96b75..90cb533d72 100644 --- a/src/components/ProviderManager.test.tsx +++ b/src/components/ProviderManager.test.tsx @@ -308,6 +308,20 @@ function mockProviderProfilesModule(options?: { } }, getProviderProfiles: options?.getProviderProfiles ?? (() => []), + // Matches ProviderManager formSteps / persistDraft / profileSummary import. + // Real helper is openai-compatibility-mode; mirror that for test presets. + providerProfileSupportsSelfHostedTools: (provider: string) => + ![ + 'anthropic', + 'custom-anthropic', + 'gemini', + 'mistral', + 'github', + 'github-enterprise', + 'bedrock', + 'vertex', + 'minimax', + ].includes(provider), setActiveProviderProfile: options?.setActiveProviderProfile ?? (() => null), updateProviderProfile: options?.updateProviderProfile ?? (() => null), })) @@ -697,6 +711,10 @@ test('ProviderManager shows API mode picker for custom OpenAI-compatible provide frame.includes('Default model'), ) mounted.stdin.write('\r') + await waitForFrameOutput(mounted.getOutput, frame => + frame.includes('Self-hosted tools'), + ) + mounted.stdin.write('\r') const output = await waitForFrameOutput(mounted.getOutput, frame => frame.includes('API mode') && frame.includes('Automatic'), @@ -1367,13 +1385,13 @@ test('ProviderManager clears hidden Hicap auth fields when editing', async () => mounted.stdin.write('\r') await waitForFrameOutput(mounted.getOutput, frame => frame.includes('Edit provider profile') && - frame.includes('Step 1 of 6'), + frame.includes('Step 1 of 7'), ) - for (let step = 2; step <= 6; step++) { + for (let step = 2; step <= 7; step++) { mounted.stdin.write('\r') await waitForFrameOutput(mounted.getOutput, frame => - frame.includes(`Step ${step} of 6`), + frame.includes(`Step ${step} of 7`), ) } mounted.stdin.write('\r') @@ -1443,25 +1461,31 @@ test('ProviderManager skips advanced fields for legacy Kimi Code profiles', asyn await waitForFrameOutput(mounted.getOutput, frame => frame.includes('Edit provider profile') && frame.includes('Provider name') && - frame.includes('Step 1 of 4'), + frame.includes('Step 1 of 5'), ) mounted.stdin.write('\r') await waitForFrameOutput(mounted.getOutput, frame => frame.includes('Base URL') && - frame.includes('Step 2 of 4'), + frame.includes('Step 2 of 5'), ) mounted.stdin.write('\r') await waitForFrameOutput(mounted.getOutput, frame => frame.includes('Default model') && - frame.includes('Step 3 of 4'), + frame.includes('Step 3 of 5'), + ) + + mounted.stdin.write('\r') + await waitForFrameOutput(mounted.getOutput, frame => + frame.includes('Self-hosted tools') && + frame.includes('Step 4 of 5'), ) mounted.stdin.write('\r') const output = await waitForFrameOutput(mounted.getOutput, frame => frame.includes('API key') && - frame.includes('Step 4 of 4'), + frame.includes('Step 5 of 5'), ) expect(output).not.toContain('API mode') @@ -2431,50 +2455,16 @@ test('ProviderManager editing an active multi-model provider keeps app state on mounted.getOutput, frame => frame.includes('Edit provider profile') && - frame.includes('Step 1 of 8'), - ) - - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 2 of 8'), - ) - - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 3 of 8'), - ) - - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 4 of 8'), - ) - - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 5 of 8'), - ) - - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 6 of 8'), + frame.includes('Step 1 of 9'), ) - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 7 of 8'), - ) - - mounted.stdin.write('\r') - await waitForFrameOutput( - mounted.getOutput, - frame => frame.includes('Step 8 of 8'), - ) + for (let step = 2; step <= 9; step++) { + mounted.stdin.write('\r') + await waitForFrameOutput( + mounted.getOutput, + frame => frame.includes(`Step ${step} of 9`), + ) + } mounted.stdin.write('\r') diff --git a/src/components/ProviderManager.tsx b/src/components/ProviderManager.tsx index d17b0fbfb3..9e22441342 100644 --- a/src/components/ProviderManager.tsx +++ b/src/components/ProviderManager.tsx @@ -67,6 +67,7 @@ import { getActiveProviderProfile, getProviderPresetDefaults, getProviderProfiles, + providerProfileSupportsSelfHostedTools, setActiveProviderProfile, type ProviderPreset, type ProviderProfileInput, @@ -145,6 +146,7 @@ type DraftField = | 'model' | 'apiKey' | 'apiFormat' + | 'selfHostedTools' | 'authHeader' | 'authHeaderValue' | 'customHeaders' @@ -196,6 +198,14 @@ const FORM_STEPS: Array<{ placeholder: 'e.g. llama3.1:8b or glm-4.7; glm-4.7-flash', helpText: 'Model name(s) to use. Separate multiple with ";" or ","; first is default.', }, + { + key: 'selfHostedTools', + label: 'Self-hosted tools', + placeholder: 'auto', + helpText: + 'For llama-server / vLLM / Ollama on this profile only. Automatic = local auto-detect; Enabled/Disabled force recovery on or off. No shell env.', + optional: true, + }, { key: 'apiFormat', label: 'API mode', @@ -252,6 +262,12 @@ function toDraft(profile: ProviderProfile): ProviderDraft { model: profile.model, apiKey: profile.apiKey ?? '', apiFormat: profile.apiFormat ?? 'auto', + selfHostedTools: + profile.selfHostedTools === true + ? 'enabled' + : profile.selfHostedTools === false + ? 'disabled' + : 'auto', authHeader: profile.authHeader ?? '', authHeaderValue: profile.authHeaderValue ?? '', customHeaders: serializeProfileCustomHeaders(profile.customHeaders) ?? '', @@ -286,7 +302,13 @@ function presetToDraft(preset: ProviderPreset): ProviderDraft { baseUrl: defaults.baseUrl, model: defaults.model, apiKey: defaults.apiKey ?? '', - apiFormat: 'chat_completions', + apiFormat: preset === 'custom' ? 'auto' : 'chat_completions', + selfHostedTools: + defaults.selfHostedTools === true + ? 'enabled' + : defaults.selfHostedTools === false + ? 'disabled' + : 'auto', authHeader: '', authHeaderValue: '', customHeaders: '', @@ -321,11 +343,16 @@ function profileSummary(profile: ProviderProfile, isActive: boolean): string { routeSupportsApiFormatSelection(routeId) ? ` · ${profile.apiFormat === 'responses_compat' ? 'responses (compat)' : profile.apiFormat === 'responses' ? 'responses' : profile.apiFormat === 'chat_completions' ? 'chat/completions' : 'automatic'}` : '' + const selfHostedInfo = + providerProfileSupportsSelfHostedTools(profile.provider) && + profile.selfHostedTools + ? ' · self-hosted tools' + : '' const authInfo = routeSupportsAuthHeaders(routeId) && profile.authHeader ? ` · ${profile.authHeader} auth` : '' - return `${providerKind} · ${profile.baseUrl} · ${modelDisplay}${modeInfo}${authInfo} · ${keyInfo}${activeSuffix}` + return `${providerKind} · ${profile.baseUrl} · ${modelDisplay}${modeInfo}${selfHostedInfo}${authInfo} · ${keyInfo}${activeSuffix}` } function getGithubCredentialSourceFromEnv( @@ -890,6 +917,9 @@ export function ProviderManager({ mode, onDone }: Props): React.ReactNode { const showsAuthHeaderValue = routeShowsAuthHeaderValue(routeId) const showsCustomHeaders = routeShowsCustomHeaders(routeId) return FORM_STEPS.filter(step => { + if (step.key === 'selfHostedTools') { + return providerProfileSupportsSelfHostedTools(draftProvider) + } if (step.key === 'apiFormat') { return routeSupportsApiFormatSelection(routeId) } @@ -1576,16 +1606,8 @@ export function ProviderManager({ mode, onDone }: Props): React.ReactNode { function startCreateFromPreset(preset: ProviderPreset): void { const defaults = getProviderPresetDefaults(preset) const provider = defaults.provider ?? 'openai' - const nextDraft = { - name: defaults.name, - baseUrl: defaults.baseUrl, - model: defaults.model, - apiKey: defaults.apiKey ?? '', - apiFormat: preset === 'custom' ? 'auto' : 'chat_completions', - authHeader: '', - authHeaderValue: '', - customHeaders: '', - } + const nextDraft = presetToDraft(preset) + nextDraft.apiKey = defaults.apiKey ?? '' setEditingProfileId(null) setDraftProvider(provider) setDraft(nextDraft) @@ -1706,6 +1728,13 @@ export function ProviderManager({ mode, onDone }: Props): React.ReactNode { Object.keys(parsedCustomHeaders.headers).length > 0 ? parsedCustomHeaders.headers : undefined, + selfHostedTools: providerProfileSupportsSelfHostedTools(provider) + ? nextDraft.selfHostedTools === 'enabled' + ? true + : nextDraft.selfHostedTools === 'disabled' + ? false + : undefined + : undefined, } const saved = profileId @@ -2207,7 +2236,43 @@ export function ProviderManager({ mode, onDone }: Props): React.ReactNode { Step {formStepIndex + 1} of {formSteps.length}: {displayStep.label} - {currentStepKey === 'apiFormat' ? ( + {currentStepKey === 'selfHostedTools' ? ( + { expect(env.GEMINI_API_KEY).toBe('gemini-key') expect(env.ANTHROPIC_API_KEY).toBe('anthropic-key') }) + + test('clears parent self-hosted recovery flags for the override route', () => { + const env: Record = { + CLAUDE_CODE_USE_OPENAI: '1', + OPENAI_BASE_URL: 'https://llama.example.com:8443/v1', + OPENAI_MODEL: 'qwen-parent', + OPENAI_SELF_HOSTED_TOOLS: '1', + OPENAI_PARSE_TEXT_TOOL_CALLS: '1', + OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS: '1', + OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS: '1', + } + + applyAgentProviderOverrideToEnv( + { + model: 'gpt-4o', + baseURL: 'https://api.openai.com/v1', + apiKey: 'sk-oai', + }, + env, + ) + + expect(env.CLAUDE_CODE_USE_OPENAI).toBe('1') + expect(env.OPENAI_BASE_URL).toBe('https://api.openai.com/v1') + expect(env.OPENAI_MODEL).toBe('gpt-4o') + expect(env.OPENAI_SELF_HOSTED_TOOLS).toBeUndefined() + expect(env.OPENAI_PARSE_TEXT_TOOL_CALLS).toBeUndefined() + expect(env.OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS).toBeUndefined() + expect(env.OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS).toBeUndefined() + }) }) describe('shouldEnforceModelAllowlist', () => { diff --git a/src/services/api/agentRouting.ts b/src/services/api/agentRouting.ts index 32b358e36a..a53ef36b8f 100644 --- a/src/services/api/agentRouting.ts +++ b/src/services/api/agentRouting.ts @@ -57,6 +57,11 @@ const PROVIDER_ENV_VARS_TO_CLEAR_FOR_OVERRIDE = [ 'OPENAI_AUTH_HEADER', 'OPENAI_AUTH_SCHEME', 'OPENAI_AUTH_HEADER_VALUE', + // Parent profile self-hosted recovery must not apply to the override route. + 'OPENAI_SELF_HOSTED_TOOLS', + 'OPENAI_PARSE_TEXT_TOOL_CALLS', + 'OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS', + 'OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS', ] as const /** diff --git a/src/services/api/openaiShim.ollamaTextToolCalls.test.ts b/src/services/api/openaiShim.ollamaTextToolCalls.test.ts index 90846921bd..9728819799 100644 --- a/src/services/api/openaiShim.ollamaTextToolCalls.test.ts +++ b/src/services/api/openaiShim.ollamaTextToolCalls.test.ts @@ -26,6 +26,21 @@ type OpenAIShimClient = { } } + +const SAMPLE_TOOLS = [ + { + name: 'Bash', + description: 'run shell', + input_schema: { type: 'object', properties: {} }, + }, + { + name: 'Read', + description: 'read file', + input_schema: { type: 'object', properties: {} }, + }, +] as const + + function makeOllamaNativeStreamingResponse(lines: string[]): Response { const encoder = new TextEncoder() return new Response( @@ -117,6 +132,107 @@ describe('parseTextToolCalls', () => { expect(calls).toHaveLength(0) }) + test('ignores person/package-style {"name":...} without arguments field', () => { + const { calls } = parseTextToolCalls( + 'Here is an example person object:\n{"name":"Alice","age":30}', + ) + expect(calls).toHaveLength(0) + }) + + test('rejects malformed JSON string arguments (no silent {})', () => { + const { calls } = parseTextToolCalls( + '{"name":"Bash","arguments":"{not-json"}', + ) + expect(calls).toHaveLength(0) + }) + + test('rejects array arguments value', () => { + const { calls } = parseTextToolCalls( + '{"name":"Bash","arguments":["ls","-la"]}', + ) + expect(calls).toHaveLength(0) + }) + + test('rejects arguments string "null" (not empty object)', () => { + const { calls } = parseTextToolCalls( + '{"name":"Bash","arguments":"null"}', + ) + expect(calls).toHaveLength(0) + }) + + test('without advertised tools, Ollama path does not recover JSON as tool_use', async () => { + // Gate requires tools — a normal Ollama JSON-looking answer must stay text. + const previousFetch = globalThis.fetch + const originalOpenAIBaseUrl = process.env.OPENAI_BASE_URL + const originalOpenAIApiKey = process.env.OPENAI_API_KEY + process.env.OPENAI_BASE_URL = 'http://localhost:11434/v1' + process.env.OPENAI_API_KEY = 'test-key' + globalThis.fetch = (async () => + makeOllamaNativeStreamingResponse( + makeNdjsonChunks([ + ollamaChunk('{"name":"status","arguments":{"ok":true}}'), + ollamaChunk('', 'stop'), + ]), + )) as unknown as FetchType + try { + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result = await client.beta.messages + .create({ + model: 'qwen2.5:7b', + messages: [{ role: 'user', content: 'status?' }], + max_tokens: 64, + stream: true, + }) + .withResponse() + const events: Record[] = [] + for await (const event of result.data) events.push(event) + const toolStarts = events.filter( + e => + e.type === 'content_block_start' && + (e.content_block as Record)?.type === 'tool_use', + ) + expect(toolStarts).toHaveLength(0) + const text = events + .filter( + e => + e.type === 'content_block_delta' && + (e.delta as Record)?.type === 'text_delta', + ) + .map(e => (e.delta as Record).text) + .join('') + expect(text).toContain('"name":"status"') + } finally { + globalThis.fetch = previousFetch + if (originalOpenAIBaseUrl === undefined) { + delete process.env.OPENAI_BASE_URL + } else { + process.env.OPENAI_BASE_URL = originalOpenAIBaseUrl + } + if (originalOpenAIApiKey === undefined) { + delete process.env.OPENAI_API_KEY + } else { + process.env.OPENAI_API_KEY = originalOpenAIApiKey + } + } + }) + + test('rejects name not in allowedToolNames allowlist', () => { + const text = '{"name":"Bash","arguments":{"command":"ls"}}' + const { calls } = parseTextToolCalls(text, { + allowedToolNames: new Set(['Read', 'Write']), + }) + expect(calls).toHaveLength(0) + }) + + test('accepts name present in allowedToolNames allowlist', () => { + const text = '{"name":"Bash","arguments":{"command":"ls"}}' + const { calls } = parseTextToolCalls(text, { + allowedToolNames: new Set(['Bash', 'Read']), + }) + expect(calls).toHaveLength(1) + expect(calls[0].name).toBe('Bash') + }) + // P1 context guard — bare JSON followed by explanatory prose must not be extracted test('skips bare JSON immediately followed by explanatory text (P1 guard)', () => { const text = @@ -252,6 +368,7 @@ describe('Ollama streaming — think-tag filtering on text-tool fallback (P1)', .create({ model: 'qwen2.5:7b', messages: [{ role: 'user', content: 'run ls' }], + tools: [...SAMPLE_TOOLS], max_tokens: 64, stream: true, }) @@ -416,6 +533,7 @@ describe('Ollama streaming — visible text before real structured tool_calls (P .create({ model: 'qwen2.5:7b', messages: [{ role: 'user', content: 'run ls' }], + tools: [...SAMPLE_TOOLS], max_tokens: 64, stream: true, }) @@ -471,6 +589,7 @@ describe('Ollama streaming — visible prose before text-based tool-call fallbac .create({ model: 'qwen2.5:7b', messages: [{ role: 'user', content: 'read the file' }], + tools: [...SAMPLE_TOOLS], max_tokens: 64, stream: true, }) @@ -568,6 +687,7 @@ describe('Ollama streaming — non-stop terminal finish reasons flush buffer', ( .create({ model: 'qwen2.5:7b', messages: [{ role: 'user', content: 'run ls' }], + tools: [...SAMPLE_TOOLS], max_tokens: 8, stream: true, }) @@ -653,4 +773,58 @@ describe('Ollama streaming — non-stop terminal finish reasons flush buffer', ( const delta = messageDelta?.delta as Record | undefined expect(delta?.stop_reason).not.toBe('tool_use') }) + + test('text-form tool recovered when finish_reason is tool_calls without delta.tool_calls', async () => { + // Compat servers may mark text-form tools with finish_reason tool_calls + // without ever emitting structured delta.tool_calls. Recovery must still + // run before the terminal delta, or stop_reason becomes tool_use with no + // tool block and the buffer flushes as plain text at EOF. + globalThis.fetch = (async () => + makeOllamaNativeStreamingResponse( + makeNdjsonChunks([ + ollamaChunk('{"name":"Bash","arguments":{"command":"pwd"}}'), + ollamaChunk('', 'tool_calls'), + ]), + )) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result = await client.beta.messages + .create({ + model: 'qwen2.5:7b', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [...SAMPLE_TOOLS], + max_tokens: 64, + stream: true, + }) + .withResponse() + + const events: Record[] = [] + for await (const event of result.data) events.push(event) + + const toolStarts = events.filter( + e => + e.type === 'content_block_start' && + (e.content_block as Record)?.type === 'tool_use', + ) + expect(toolStarts).toHaveLength(1) + expect((toolStarts[0].content_block as Record).name).toBe( + 'Bash', + ) + + const allText = events + .filter( + e => + e.type === 'content_block_delta' && + (e.delta as Record)?.type === 'text_delta', + ) + .map(e => (e.delta as Record).text) + .join('') + expect(allText).not.toContain('{"name":"Bash"') + + const messageDelta = events.find(e => e.type === 'message_delta') as + | Record + | undefined + const delta = messageDelta?.delta as Record | undefined + expect(delta?.stop_reason).toBe('tool_use') + }) }) diff --git a/src/services/api/openaiShim.test.ts b/src/services/api/openaiShim.test.ts index 0f4278b266..28f0d8daf4 100644 --- a/src/services/api/openaiShim.test.ts +++ b/src/services/api/openaiShim.test.ts @@ -8616,11 +8616,264 @@ test('injects semantic assistant message when tool result is followed by user me }) // openaiShim test extraction seam 136 end +test('injects semantic boundary for Mistral models', async () => { + // Mistral API requires tool → assistant placeholder → user pattern with + // "[Tool results received]" assistant message before user content. + process.env.OPENAI_BASE_URL = 'https://api.mistral.ai/v1' + process.env.OPENAI_API_KEY = 'sk-mistral-test' + + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + return new Response( + JSON.stringify({ + id: 'chatcmpl-mistral', + object: 'chat.completion', + created: 123456789, + model: 'mistral-large-latest', + choices: [ + { message: { role: 'assistant', content: 'ok' }, finish_reason: 'stop' }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { headers: { 'Content-Type': 'application/json' } }, + ) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + + await client.beta.messages.create({ + model: 'mistral-large-latest', + messages: [ + { + role: 'assistant', + content: [{ type: 'tool_use', id: 'call_1', name: 'Bash', input: { command: 'ls' } }], + }, + { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'call_1', content: 'file.txt' }], + }, + { role: 'user', content: 'What is in the directory?' }, + ], + max_tokens: 64, + stream: false, + }) + + const messages = requestBody?.messages as Array> + const roles = messages.map(m => m.role) + // Mistral path should inject assistant placeholder between tool_result and user + expect(roles).toEqual(['assistant', 'tool', 'assistant', 'user']) + const assistantPlaceholder = messages.find( + m => m.role === 'assistant' && m.content === '[Tool results received]', + ) + expect(assistantPlaceholder).toBeDefined() +}) + +test('does not inject semantic boundary for llama-server / non-Mistral models', async () => { + // llama-server + Qwen (or any non-Mistral OpenAI-compat): tool → user must + // stay intact. Injecting "[Tool results received]" makes the model echo it + // and stall after tools. + process.env.OPENAI_BASE_URL = 'http://127.0.0.1:8080/v1' + process.env.OPENAI_API_KEY = 'none' + + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + return new Response( + JSON.stringify({ + id: 'chatcmpl-llama', + object: 'chat.completion', + created: 123456789, + model: 'qwen3.6:35b', + choices: [ + { message: { role: 'assistant', content: 'ok' }, finish_reason: 'stop' }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { headers: { 'Content-Type': 'application/json' } }, + ) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + + await client.beta.messages.create({ + model: 'qwen3.6:35b', + messages: [ + { + role: 'assistant', + content: [{ type: 'tool_use', id: 'call_1', name: 'Bash', input: { command: 'ls' } }], + }, + { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'call_1', content: 'file.txt' }], + }, + { role: 'user', content: 'What is in the directory?' }, + ], + max_tokens: 64, + stream: false, + }) + + const messages = requestBody?.messages as Array> + const roles = messages.map(m => m.role) + expect(roles).toEqual(['assistant', 'tool', 'user']) + expect( + messages.some( + m => + m.role === 'assistant' && m.content === '[Tool results received]', + ), + ).toBe(false) +}) + +test('does not inject semantic boundary for public OpenAI-compatible models', async () => { + process.env.OPENAI_BASE_URL = 'https://api.openai.com/v1' + process.env.OPENAI_API_KEY = 'sk-test' + + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + return new Response( + JSON.stringify({ + id: 'chatcmpl-oai', + object: 'chat.completion', + created: 123456789, + model: 'gpt-4o', + choices: [ + { message: { role: 'assistant', content: 'ok' }, finish_reason: 'stop' }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { headers: { 'Content-Type': 'application/json' } }, + ) + }) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + + await client.beta.messages.create({ + model: 'gpt-4o', + messages: [ + { + role: 'assistant', + content: [{ type: 'tool_use', id: 'call_1', name: 'search', input: {} }], + }, + { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'call_1', content: 'Result' }], + }, + { role: 'user', content: 'Next' }, + ], + max_tokens: 64, + stream: false, + }) + + const roles = (requestBody?.messages as Array>).map( + m => m.role, + ) + expect(roles).toEqual(['assistant', 'tool', 'user']) +}) + +test('non-stream recovers JSON-in-text tool calls (llama-server style)', async () => { + process.env.OPENAI_BASE_URL = 'http://192.168.1.10:8080/v1' + process.env.OPENAI_API_KEY = 'none' + + globalThis.fetch = (async () => + new Response( + JSON.stringify({ + id: 'chatcmpl-text-tools', + object: 'chat.completion', + created: 123456789, + model: 'qwen3.6:35b', + choices: [ + { + message: { + role: 'assistant', + content: + 'Checking.\n{"name":"Bash","arguments":{"command":"pwd"}}', + }, + finish_reason: 'stop', + }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { headers: { 'Content-Type': 'application/json' } }, + )) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result = (await client.beta.messages.create({ + model: 'qwen3.6:35b', + messages: [{ role: 'user', content: 'run pwd' }], + // Tools must be advertised — recovery is gated like the stream path. + tools: [ + { + name: 'Bash', + description: 'run shell', + input_schema: { type: 'object', properties: {} }, + }, + ], + max_tokens: 64, + stream: false, + })) as { + content: Array<{ type: string; name?: string; text?: string; input?: unknown }> + stop_reason: string + } + + expect(result.stop_reason).toBe('tool_use') + const tool = result.content.find(b => b.type === 'tool_use') + expect(tool?.name).toBe('Bash') + expect(tool?.input).toEqual({ command: 'pwd' }) + const text = result.content.find(b => b.type === 'text') + expect(text?.text).toContain('Checking') + expect(text?.text).not.toContain('"name":"Bash"') +}) + +test('non-stream does not treat cloud prose ending in {"name":...} as tool_use', async () => { + // Regression: ungated parseTextToolCalls converted example JSON objects + // (person records, package.json fragments) into phantom tool_use on cloud. + process.env.OPENAI_BASE_URL = 'https://api.openai.com/v1' + process.env.OPENAI_API_KEY = 'sk-test' + + globalThis.fetch = (async () => + new Response( + JSON.stringify({ + id: 'chatcmpl-cloud-prose', + object: 'chat.completion', + created: 123456789, + model: 'gpt-4o', + choices: [ + { + message: { + role: 'assistant', + content: + 'Here is an example person object:\n\n```json\n{"name": "Alice", "age": 30}\n```', + }, + finish_reason: 'stop', + }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { headers: { 'Content-Type': 'application/json' } }, + )) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result = (await client.beta.messages.create({ + model: 'gpt-4o', + messages: [{ role: 'user', content: 'show an example' }], + max_tokens: 64, + stream: false, + })) as { + content: Array<{ type: string; name?: string; text?: string; input?: unknown }> + stop_reason: string + } + + expect(result.stop_reason).toBe('end_turn') + expect(result.content).toHaveLength(1) + expect(result.content[0]?.type).toBe('text') + expect(result.content[0]?.text).toContain('"name": "Alice"') + expect(result.content.some(b => b.type === 'tool_use')).toBe(false) +}) -// Extraction boundary: executor tool self-healing | message/provider shaping. -// Provider request shaping below is not owned by the executor. -// Keep this marker stable for independent adjacent test migrations. -// openaiShim test extraction seam 137 start: Moonshot: uses max_tokens (not max_completion_tokens) and strips store test('Moonshot: uses max_tokens (not max_completion_tokens) and strips store', async () => { process.env.OPENAI_BASE_URL = 'https://api.moonshot.ai/v1' process.env.OPENAI_API_KEY = 'sk-moonshot-test' diff --git a/src/services/api/openaiShim.ts b/src/services/api/openaiShim.ts index 58ddeb27b1..5f92f60b5d 100644 --- a/src/services/api/openaiShim.ts +++ b/src/services/api/openaiShim.ts @@ -102,6 +102,9 @@ import { resolveRuntimeCodexCredentials, resolveProviderRequest, shouldAttemptLocalToollessRetry, + shouldInjectToolResultSemanticBoundary, + shouldUseSelfHostedToolCompat, + TOOL_RESULT_SEMANTIC_PLACEHOLDER, type LocalFastPathConfig, } from './providerConfig.js' import { @@ -162,6 +165,9 @@ import { import { convertMessages as convertAnthropicMessages, convertSystemPrompt as convertSystemPromptImpl, + convertContentBlocks, + convertToolResultContent, + joinTextContentParts, } from './openaiShim/messageConversion.js' const GITHUB_429_MAX_RETRIES = 3 @@ -736,14 +742,326 @@ function convertMessages( reasoningContentFallback?: '' | 'omit' preserveGeminiThoughtSignature?: boolean supportsImageInputs?: boolean + injectToolResultSemanticBoundary?: boolean }, ): OpenAIMessage[] { - return convertAnthropicMessages(messages, system, { - ...options, - getGeminiThoughtSignature: geminiThoughtSignatureFromExtraContent, - mergeGeminiThoughtSignature, - log: message => logForDebugging(message), - }) + const preserveReasoningContent = options?.preserveReasoningContent === true + const reasoningContentFallback = options?.reasoningContentFallback + const preserveGeminiThoughtSignature = options?.preserveGeminiThoughtSignature === true + const supportsImageInputs = options?.supportsImageInputs + const injectToolResultSemanticBoundary = + options?.injectToolResultSemanticBoundary === true + const result: OpenAIMessage[] = [] + const knownToolCallIds = new Set() + + // Pre-scan for all tool results in the history to identify valid tool calls + const toolResultIds = new Set() + for (const msg of messages) { + const inner = msg.message ?? msg + const content = (inner as { content?: unknown }).content + if (Array.isArray(content)) { + for (const block of content) { + if ( + (block as { type?: string }).type === 'tool_result' && + (block as { tool_use_id?: string }).tool_use_id + ) { + toolResultIds.add((block as { tool_use_id: string }).tool_use_id) + } + } + } + } + + // System message first + const sysText = convertSystemPrompt(system) + if (sysText) { + result.push({ role: 'system', content: sysText }) + } + + for (let i = 0; i < messages.length; i++) { + const msg = messages[i] + const isLastInHistory = i === messages.length - 1 + + // Claude Code wraps messages in { role, message: { role, content } } + const inner = msg.message ?? msg + const role = (inner as { role?: string }).role ?? msg.role + const content = (inner as { content?: unknown }).content + + if (role === 'user') { + // Check for tool_result blocks in user messages + if (Array.isArray(content)) { + let otherContent: unknown[] | undefined + + // Emit tool results as tool messages, but ONLY if we have a matching tool_use ID. + // Mistral/OpenAI strictly require tool messages to follow an assistant message with tool_calls. + // If the user interrupted (ESC) and a synthetic tool_result was generated without a recorded tool_use, + // emitting it here would cause a "role must alternate" or "unexpected role" error. + for (const block of content) { + const blockType = (block as { type?: string }).type + if (blockType === 'tool_result') { + const tr = block as { + tool_use_id?: string + content?: unknown + is_error?: boolean + } + const id = tr.tool_use_id ?? 'unknown' + if (knownToolCallIds.has(id)) { + result.push({ + role: 'tool', + tool_call_id: id, + content: convertToolResultContent(tr.content, tr.is_error, { supportsImageInputs }), + }) + } else { + logForDebugging( + `Dropping orphan tool_result for ID: ${id} to prevent API error`, + ) + } + } else { + otherContent ??= [] + otherContent.push(block) + } + } + + // Emit remaining user content + if (otherContent && otherContent.length > 0) { + result.push({ + role: 'user', + content: convertContentBlocks(otherContent, { supportsImageInputs }), + }) + } + } else { + result.push({ + role: 'user', + content: convertContentBlocks(content, { supportsImageInputs }), + }) + } + } else if (role === 'assistant') { + // Check for tool_use blocks + if (Array.isArray(content)) { + let toolUses: Array<{ + id?: string + name?: string + input?: unknown + extra_content?: Record + signature?: string + }> | undefined + let thinkingBlock: + | { type?: string; thinking?: string; data?: string; signature?: string } + | undefined + let textContent: unknown[] | undefined + + for (const block of content) { + const blockType = (block as { type?: string }).type + if (blockType === 'tool_use') { + toolUses ??= [] + toolUses.push( + block as { + id?: string + name?: string + input?: unknown + extra_content?: Record + signature?: string + }, + ) + } else if ( + blockType === 'thinking' || + blockType === 'redacted_thinking' + ) { + thinkingBlock ??= block as { + type?: string + thinking?: string + data?: string + signature?: string + } + } else { + textContent ??= [] + textContent.push(block) + } + } + + const assistantMsg: OpenAIMessage = { + role: 'assistant', + content: (() => { + const c = convertContentBlocks(textContent ?? [], { supportsImageInputs }) + return typeof c === 'string' + ? c + : Array.isArray(c) + ? joinTextContentParts(c) + : '' + })(), + } + + // Providers that validate reasoning continuity (Moonshot/Kimi Code: "thinking + // is enabled but reasoning_content is missing in assistant tool call + // message at index N" 400) need the original chain-of-thought echoed + // back on each assistant message that carries a tool_call. We kept + // the thinking block on the Anthropic side; re-attach it here as the + // `reasoning_content` field on the outgoing OpenAI-shaped message. + // Gated per-provider because other endpoints either ignore the field + // (harmless) or strict-reject unknown fields (harmful). + if (preserveReasoningContent) { + // `thinking` blocks carry their content in `.thinking`; `redacted_thinking` + // blocks carry it in `.data` (see token estimation and message-size + // accounting). Read the right field per type so a real redacted block + // with non-empty content is not silently dropped to "". + const thinkingText = + thinkingBlock?.type === 'redacted_thinking' + ? thinkingBlock?.data + : thinkingBlock?.thinking + if (typeof thinkingText === 'string' && thinkingText.trim().length > 0) { + assistantMsg.reasoning_content = thinkingText + } else if ( + (toolUses?.length ?? 0) > 0 && + reasoningContentFallback === '' + ) { + assistantMsg.reasoning_content = '' + } + } + + if (toolUses && toolUses.length > 0) { + const mappedToolCalls: NonNullable = [] + for (const tu of toolUses) { + const id = tu.id ?? `call_${crypto.randomUUID().replace(/-/g, '')}` + + // Only keep tool calls that have a corresponding result in the history, + // or if it's the last message (prefill scenario). + // Orphaned tool calls (e.g. from user interruption) cause 400 errors. + if (!toolResultIds.has(id) && !isLastInHistory) { + continue + } + + knownToolCallIds.add(id) + const toolCall: NonNullable< + OpenAIMessage['tool_calls'] + >[number] = { + id, + type: 'function' as const, + function: { + name: tu.name ?? 'unknown', + arguments: + typeof tu.input === 'string' + ? tu.input + : JSON.stringify(tu.input ?? {}), + }, + } + + // Preserve existing extra_content if present + if (tu.extra_content) { + toolCall.extra_content = { ...tu.extra_content } + } + + // Gemini OpenAI-compatible endpoints require Google's + // thought_signature to be replayed with prior function-call + // parts. Preserve only real signatures received from the + // provider; synthetic placeholders are rejected by GMI. + if (preserveGeminiThoughtSignature) { + const signature = + tu.signature ?? + geminiThoughtSignatureFromExtraContent(tu.extra_content) ?? + thinkingBlock?.signature + + toolCall.extra_content = mergeGeminiThoughtSignature( + toolCall.extra_content, + signature, + ) + } + + mappedToolCalls.push(toolCall) + } + + if (mappedToolCalls.length > 0) { + assistantMsg.tool_calls = mappedToolCalls + } + } + + // Only push assistant message if it has content or tool calls. + // Stripped thinking-only blocks from user interruptions are empty and cause 400s. + if (assistantMsg.content || assistantMsg.tool_calls?.length) { + result.push(assistantMsg) + } + } else { + const assistantMsg: OpenAIMessage = { + role: 'assistant', + content: (() => { + const c = convertContentBlocks(content, { supportsImageInputs }) + return typeof c === 'string' + ? c + : Array.isArray(c) + ? joinTextContentParts(c) + : '' + })(), + } + + if (assistantMsg.content) { + result.push(assistantMsg) + } + } + } + } + + // Coalescing pass: merge consecutive messages of the same role. + // OpenAI/vLLM/Ollama require strict user↔assistant alternation. + // Multiple consecutive tool messages are allowed (assistant → tool* → user). + // Consecutive user or assistant messages must be merged to avoid Jinja + // template errors like "roles must alternate" (Devstral, Mistral models). + const coalesced: OpenAIMessage[] = [] + for (const msg of result) { + const prev = coalesced[coalesced.length - 1] + + // Mistral/Devstral only: 'tool' must be followed by 'assistant' before + // 'user'. llama-server / Qwen / Ollama / OpenAI must NOT get this — + // the synthetic assistant content is echoed as a real reply and stalls + // the tool loop (visible "[Tool results received]" in the UI). + if ( + injectToolResultSemanticBoundary && + prev && + prev.role === 'tool' && + msg.role === 'user' + ) { + coalesced.push({ + role: 'assistant', + content: TOOL_RESULT_SEMANTIC_PLACEHOLDER, + }) + } + + const lastAfterPossibleInjection = coalesced[coalesced.length - 1] + if ( + lastAfterPossibleInjection && + lastAfterPossibleInjection.role === msg.role && + msg.role !== 'tool' && + msg.role !== 'system' + ) { + const prevContent = lastAfterPossibleInjection.content + const curContent = msg.content + + if (typeof prevContent === 'string' && typeof curContent === 'string') { + lastAfterPossibleInjection.content = + prevContent + (prevContent && curContent ? '\n' : '') + curContent + } else { + const toArray = ( + c: string | OpenAIContentPart[] | undefined, + ): OpenAIContentPart[] => { + if (!c) return [] + if (typeof c === 'string') return c ? [{ type: 'text', text: c }] : [] + return c + } + lastAfterPossibleInjection.content = [ + ...toArray(prevContent), + ...toArray(curContent), + ] + } + + if (msg.tool_calls?.length) { + lastAfterPossibleInjection.tool_calls = [ + ...(lastAfterPossibleInjection.tool_calls ?? []), + ...msg.tool_calls, + ] + } + } else { + coalesced.push(msg) + } + } + + return coalesced } function getChatMessagesForTransport( transport: string, @@ -1039,46 +1357,84 @@ function extractBalancedJson(text: string, start: number): string | null { return null } +/** + * Decode a tool-call `arguments` value to a plain object. + * Rejects arrays, primitives, and malformed JSON strings (no silent `{}`). + * `null` / `undefined` → empty object (explicit empty args). + */ +function decodeToolCallArguments( + rawArgs: unknown, +): Record | null { + if (rawArgs === undefined || rawArgs === null) { + return {} + } + if (typeof rawArgs === 'string') { + try { + rawArgs = JSON.parse(rawArgs) + } catch { + return null + } + } + if ( + rawArgs && + typeof rawArgs === 'object' && + !Array.isArray(rawArgs) + ) { + return rawArgs as Record + } + return null +} + function parseAndAdd( raw: string, results: ParsedTextToolCall[], seen: Set, + allowedToolNames?: ReadonlySet, ): boolean { - let obj: Record + let parsed: unknown try { - obj = JSON.parse(raw) + parsed = JSON.parse(raw) } catch { return false } + // Reject null / primitives / arrays — only plain objects are tool envelopes. + if ( + !parsed || + typeof parsed !== 'object' || + Array.isArray(parsed) + ) { + return false + } + const obj = parsed as Record let name: string | undefined - let args: Record = {} + let args: Record | null = null if (typeof obj['name'] === 'string') { - // {"name": "X", "arguments": {...}} + // Require a real tool-call shape: {"name":"X","arguments":...}. + // Person records / package.json {"name":"Alice"} must not match. + if (!('arguments' in obj)) { + return false + } name = obj['name'] as string - args = (obj['arguments'] as Record) ?? {} + args = decodeToolCallArguments(obj['arguments']) } else if ( obj['type'] === 'function' && - typeof (obj['function'] as any)?.name === 'string' + typeof (obj['function'] as { name?: unknown } | undefined)?.name === 'string' ) { // {"type":"function","function":{"name":"X","arguments":{...}}} const fn = obj['function'] as { name: string; arguments?: unknown } + if (!('arguments' in fn)) { + return false + } name = fn.name - const rawArgs = fn.arguments - args = - typeof rawArgs === 'string' - ? (() => { - try { - return JSON.parse(rawArgs) - } catch { - return {} - } - })() - : (rawArgs as Record) ?? {} + args = decodeToolCallArguments(fn.arguments) } - if (!name) return false + if (!name || args === null) return false + if (allowedToolNames && allowedToolNames.size > 0 && !allowedToolNames.has(name)) { + return false + } const dedupKey = `${name}:${JSON.stringify(args)}` if (seen.has(dedupKey)) return false @@ -1100,11 +1456,37 @@ function stripRanges(text: string, ranges: Array<[number, number]>): string { return result + text.slice(pos) } +/** Collect advertised tool names from Anthropic-shaped `tools` on shim create. */ +function toolNamesFromShimParams(tools: unknown): Set | undefined { + if (!Array.isArray(tools) || tools.length === 0) { + return undefined + } + const names = new Set() + for (const tool of tools) { + if ( + tool && + typeof tool === 'object' && + typeof (tool as { name?: unknown }).name === 'string' + ) { + names.add((tool as { name: string }).name) + } + } + return names.size > 0 ? names : undefined +} + /** Exported for unit testing only. */ -export function parseTextToolCalls(text: string): { +export function parseTextToolCalls( + text: string, + options?: { allowedToolNames?: ReadonlySet | readonly string[] }, +): { calls: ParsedTextToolCall[] toolCallRanges: Array<[number, number]> } { + const allowedToolNames = options?.allowedToolNames + ? options.allowedToolNames instanceof Set + ? options.allowedToolNames + : new Set(options.allowedToolNames) + : undefined const results: ParsedTextToolCall[] = [] const seen = new Set() const fencedRanges: Array<[number, number]> = [] @@ -1124,7 +1506,7 @@ export function parseTextToolCalls(text: string): { if (after.length > 0 && !after.startsWith('{')) continue const range: [number, number] = [match.index!, match.index! + match[0].length] fencedRanges.push(range) - if (raw && parseAndAdd(raw, results, seen)) { + if (raw && parseAndAdd(raw, results, seen, allowedToolNames)) { acceptedRanges.push(range) } } @@ -1144,7 +1526,7 @@ export function parseTextToolCalls(text: string): { if (after.length > 0 && !after.startsWith('{')) continue const range: [number, number] = [start, start + raw.length] processedRanges.push(range) - if (parseAndAdd(raw, results, seen)) { + if (parseAndAdd(raw, results, seen, allowedToolNames)) { acceptedRanges.push(range) } } @@ -1160,6 +1542,20 @@ function nextTextToolCallSequence(): number { return ++textToolCallSequence } +// --------------------------------------------------------------------------- +// Helpers for text tool call parsing +// --------------------------------------------------------------------------- + +/** Normalize allowedToolNames to a Set or undefined. */ +function normalizeAllowedToolNames( + allowedToolNames?: ReadonlySet | readonly string[], +): ReadonlySet | undefined { + if (!allowedToolNames) return undefined + return allowedToolNames instanceof Set + ? allowedToolNames + : new Set(allowedToolNames) +} + // --------------------------------------------------------------------------- // XML tool call parser (GLM / Qwen / DeepSeek family) // @@ -1290,17 +1686,46 @@ function findXmlToolCallOpener(text: string, allowHy3: boolean): number { /** Exported for unit testing only. */ export function parseXmlToolCalls(text: string, allowHy3 = false): { calls: ParsedTextToolCall[] + /** 1:1 with `calls` (unique tool invocations only). */ toolCallRanges: Array<[number, number]> + /** + * Every matching XML block range, including duplicates of the same call. + * Use for stripRanges so repeated identical tool XML does not remain visible. + */ + stripToolCallRanges: Array<[number, number]> + /** + * Parallel to `stripToolCallRanges`. Each entry is the index into `toolCallRanges` + * (and `calls`) that owns the corresponding strip range. Duplicates of the same + * unique call share the same owner index. + */ + stripRangeOwnerIndex: Array } { const results: ParsedTextToolCall[] = [] - const seen = new Set() + const seen = new Map() const ranges: Array<[number, number]> = [] - - const addCall = (name: string, args: Record) => { + const stripRangesAll: Array<[number, number]> = [] + const stripRangeOwners: Array = [] + + const addCall = ( + name: string, + args: Record, + range: [number, number], + ): boolean => { + // Always retain the block for stripping, even when the call is deduped. + stripRangesAll.push(range) const dedupKey = `${name}:${JSON.stringify(args)}` - if (seen.has(dedupKey)) return - seen.add(dedupKey) + if (seen.has(dedupKey)) { + // Duplicate — owner is the existing call index. + stripRangeOwners.push(seen.get(dedupKey)!) + return false + } + // New unique call — owner is the current results length. + stripRangeOwners.push(results.length) + seen.set(dedupKey, results.length) results.push({ id: `xml_tc_${nextTextToolCallSequence()}`, name, arguments: args }) + // Keep toolCallRanges 1:1 with emitted (unique) calls. + ranges.push(range) + return true } const hy3Blocks = allowHy3 @@ -1330,14 +1755,13 @@ export function parseXmlToolCalls(text: string, allowHy3 = false): { const { name, args } = block.parsed if (!name) continue const range = block.range - if (!hy3WrapperRanges.some(wrapper => wrapper[0] <= range[0] && range[1] <= wrapper[1])) { - ranges.push(range) - } - addCall(name, args) + // Prefer outer wrapper range for strip when present; keep 1:1 with calls. + const outer = hy3WrapperRanges.find( + wrapper => wrapper[0] <= range[0] && range[1] <= wrapper[1], + ) + addCall(name, args, outer ?? range) } - ranges.push(...hy3WrapperRanges) - for (const block of text.matchAll(XML_TOOL_CALL_BLOCK_RE)) { const inner = block[1] ?? '' const range: [number, number] = [ @@ -1392,11 +1816,15 @@ export function parseXmlToolCalls(text: string, allowHy3 = false): { } if (!name) continue - ranges.push(range) - addCall(name, args) + addCall(name, args, range) } - return { calls: results, toolCallRanges: ranges } + return { + calls: results, + toolCallRanges: ranges, + stripToolCallRanges: stripRangesAll, + stripRangeOwnerIndex: stripRangeOwners, + } } /** @@ -1662,11 +2090,22 @@ type NonStreamingOpenAIResponse = { * and the `application/json` fallback inside `openaiStreamToAnthropic` so both * apply the same tool-call extraction, stop-reason mapping, array-content * normalization, -tag stripping, and raw text tool-call recovery. + * + * `enableTextToolCallFallback` must match the streaming gate: Ollama always, + * or self-hosted compat only when tools were advertised. Cloud backends leave + * it false so benign `{"name": ...}` JSON in assistant prose is not converted + * into phantom tool_use blocks. */ function convertNonStreamingResponseToAnthropicMessage( data: NonStreamingOpenAIResponse, model: string, + options?: { + enableTextToolCallFallback?: boolean + allowedToolNames?: ReadonlySet | readonly string[] + }, ) { + const enableTextToolCallFallback = options?.enableTextToolCallFallback === true + const allowedToolNames = options?.allowedToolNames const choice = data.choices?.[0] const content: Array> = [] // An empty tool_calls array is still truthy; treat it as "no structured tool @@ -1688,12 +2127,12 @@ function convertNonStreamingResponseToAnthropicMessage( const appendTextOrRecoveredToolCalls = (rawText: string) => { const strippedContent = stripThinkTags(rawText) if (!hasStructuredToolCalls) { - const { calls: xmlToolCalls, toolCallRanges } = parseXmlToolCalls( - strippedContent, - isHy3Model(model), - ) + const { + calls: xmlToolCalls, + stripToolCallRanges: xmlStripRanges, + } = parseXmlToolCalls(strippedContent, isHy3Model(model)) if (xmlToolCalls.length > 0) { - const visibleText = stripRanges(strippedContent, toolCallRanges).trim() + const visibleText = stripRanges(strippedContent, xmlStripRanges).trim() if (visibleText) content.push({ type: 'text', text: visibleText }) for (const toolCall of xmlToolCalls) { content.push({ @@ -1705,6 +2144,27 @@ function convertNonStreamingResponseToAnthropicMessage( } return } + + // JSON-in-text tool calls (llama-server / Ollama / self-hosted). Gated + // the same way as the streaming path — never for cloud prose that + // happens to end with {"name": ...}. + if (enableTextToolCallFallback) { + const { calls: textToolCalls, toolCallRanges: textRanges } = + parseTextToolCalls(strippedContent, { allowedToolNames }) + if (textToolCalls.length > 0) { + const visibleText = stripRanges(strippedContent, textRanges).trim() + if (visibleText) content.push({ type: 'text', text: visibleText }) + for (const toolCall of textToolCalls) { + content.push({ + type: 'tool_use', + id: toolCall.id, + name: toolCall.name, + input: toolCall.arguments, + }) + } + return + } + } } const rawToolCalls = hasStructuredToolCalls @@ -1811,8 +2271,15 @@ async function* openaiStreamToAnthropic( response: Response, model: string, signal?: AbortSignal, + /** + * When true, buffer text until finish and recover JSON-in-text tool calls + * (self-hosted llama-server / Ollama / local OpenAI-compat with tools). + * Parameter name is historical (`isOllama`); any self-hosted tool path + * may set it. + */ isOllama = false, requestUrl?: string, + allowedToolNames?: ReadonlySet | readonly string[], ): AsyncGenerator { const messageId = makeMessageId() const allowHy3ToolCalls = isHy3Model(model) @@ -1834,13 +2301,11 @@ async function* openaiStreamToAnthropic( let lastStopReason: 'tool_use' | 'max_tokens' | 'end_turn' | null = null let hasEmittedFinalUsage = false let hasProcessedFinishReason = false - // Accumulated text for Ollama text-based tool call fallback parsing (#1053) + // Accumulated text for self-hosted text-based tool call fallback parsing let accumulatedText = '' - // Use the resolved value threaded from the call site (resolveProviderRequest) - // rather than re-reading env vars inside the generator. + // Self-hosted / Ollama: buffer text so raw tool-call JSON is not shown + // before extraction at finish_reason=stop. const isOllamaStream = isOllama - // Buffer Ollama text deltas so raw tool-call JSON is never emitted as text_delta - // before extraction at finish_reason=stop (P2 fix for #1053). let ollamaTextBuffer = '' const streamState = createStreamState() let bufferedRawToolCallsText: string | null = null @@ -1892,7 +2357,12 @@ async function* openaiStreamToAnthropic( // fallback preserves tool_calls, Anthropic stop-reason mapping, array // content normalization, -tag stripping, and raw text tool-call // recovery — then re-emit the resulting message as stream events. - const message = convertNonStreamingResponseToAnthropicMessage(parsed, model) + // `isOllama` is the same enable flag the live stream path uses for + // JSON-in-text recovery (Ollama / self-hosted tools). + const message = convertNonStreamingResponseToAnthropicMessage(parsed, model, { + enableTextToolCallFallback: isOllamaStream, + allowedToolNames, + }) yield { type: 'message_start', @@ -2387,13 +2857,21 @@ async function* openaiStreamToAnthropic( contentBlockIndex++ hasClosedThinking = true } - // Ollama text-based tool call fallback (#1053): + // Self-hosted / Ollama text-based tool call fallback (#1053): // Must run before closeActiveContentBlock so the text buffer can be flushed - // with tool-call JSON stripped (P2). Ollama models emit tool calls as raw - // JSON text; scan accumulated text on any terminal finish reason with no - // API tool calls. finish_reason is mutated to 'tool_calls' only for 'stop' - // so the JSON fallback remains scoped to normal completions. - const OLLAMA_TERMINAL_REASONS = new Set(['stop', 'length', 'content_filter', 'safety']) + // with tool-call JSON stripped (P2). Local models often emit tool calls as + // raw JSON text; scan accumulated text on any terminal finish reason with no + // API tool calls. Include finish_reason "tool_calls" so servers that mark + // text-form tools that way (without delta.tool_calls) still recover. + // finish_reason is mutated to 'tool_calls' only for 'stop' so the JSON + // fallback remains scoped to normal completions for other reasons. + const OLLAMA_TERMINAL_REASONS = new Set([ + 'stop', + 'length', + 'content_filter', + 'safety', + 'tool_calls', + ]) const isTerminalOllamaFinish = OLLAMA_TERMINAL_REASONS.has(choice.finish_reason ?? '') && activeToolCalls.size === 0 && @@ -2401,12 +2879,69 @@ async function* openaiStreamToAnthropic( const originalFinishReason = choice.finish_reason let ollamaClosedContentBlock = false if (isTerminalOllamaFinish) { - const { calls: textToolCalls, toolCallRanges } = parseTextToolCalls(accumulatedText) + // Prefer JSON-in-text; if none, recover Qwen/GLM XML tool calls from + // the buffered stream (non-Ollama path uses live XML holdback, but + // self-hosted buffering skips that detector). + let recoveredCalls: Array<{ + id: string + name: string + arguments: Record + }> = [] + let recoveredRanges: Array<[number, number]> = [] + const { calls: textToolCalls, toolCallRanges } = parseTextToolCalls( + accumulatedText, + { allowedToolNames }, + ) if (textToolCalls.length > 0) { + recoveredCalls = textToolCalls + recoveredRanges = toolCallRanges + } else { + const xmlParsed = parseXmlToolCalls( + accumulatedText, + allowHy3ToolCalls, + ) + // Drop names not in the allowlist; strip only blocks for accepted calls. + const allow = normalizeAllowedToolNames(allowedToolNames) + + // Build a set of unique call indices that pass the allowlist. + const acceptedCallIndices = new Set() + for (let i = 0; i < xmlParsed.calls.length; i++) { + const call = xmlParsed.calls[i]! + if (allow && allow.size > 0 && !allow.has(call.name)) { + continue + } + acceptedCallIndices.add(i) + } + + if (acceptedCallIndices.size > 0) { + // Include only accepted unique calls. + const filteredCalls: typeof recoveredCalls = [] + for (let i = 0; i < xmlParsed.calls.length; i++) { + if (acceptedCallIndices.has(i)) { + filteredCalls.push(xmlParsed.calls[i]!) + } + } + + // Include all strip ranges whose owner is an accepted call. + const filteredStripRanges: typeof xmlParsed.stripToolCallRanges = [] + for (let i = 0; i < xmlParsed.stripRangeOwnerIndex.length; i++) { + if (acceptedCallIndices.has(xmlParsed.stripRangeOwnerIndex[i]!)) { + filteredStripRanges.push(xmlParsed.stripToolCallRanges[i]!) + } + } + + recoveredCalls = filteredCalls + recoveredRanges = filteredStripRanges + } + } + if (recoveredCalls.length > 0) { ollamaClosedContentBlock = true - // Compute visible prose (tool-call JSON stripped, think-tags removed). - // Use accumulatedText (raw) as source because toolCallRanges are relative to it. - const stripped = stripRanges(accumulatedText, toolCallRanges).trim() + // Buffer is consumed via accumulatedText recovery — clear so EOF + // flush does not re-emit raw JSON/XML as visible text. + ollamaTextBuffer = '' + // Compute visible prose (tool-call payload stripped, think-tags removed). + // Use accumulatedText (raw) as source because ranges are relative to it. + const stripped = stripRanges(accumulatedText, recoveredRanges).trim() const strippedVisible = stripThinkTags(stripped).trim() if (hasEmittedContentStart) { // Text block was already open — emit stripped prose then close it. @@ -2420,7 +2955,7 @@ async function* openaiStreamToAnthropic( } yield* closeActiveContentBlock() } else if (strippedVisible) { - // Text was buffered (Ollama path, hasEmittedContentStart === false). + // Text was buffered (self-hosted path, hasEmittedContentStart === false). // Open a text block, emit the visible prose before the tool call, close it. throwIfStreamAborted(signal) yield { @@ -2437,7 +2972,7 @@ async function* openaiStreamToAnthropic( } yield* closeActiveContentBlock() } - for (const tc of textToolCalls) { + for (const tc of recoveredCalls) { throwIfStreamAborted(signal) const toolBlockIndex = contentBlockIndex yield { @@ -2479,22 +3014,24 @@ async function* openaiStreamToAnthropic( index: contentBlockIndex, delta: { type: 'text_delta', text: ollamaTextBuffer }, } + ollamaTextBuffer = '' } } - // XML tool-call fallback for non-Ollama OpenAI-compatible providers - // (GLM/Qwen emit `` as text). Mirror the Ollama - // path: convert buffered XML to tool_use blocks and strip the raw XML. + // XML tool-call fallback for cloud OpenAI-compatible providers + // (GLM/Qwen emit `` as text). Self-hosted + // buffering path above already handles JSON; XML still runs here + // when text was streamed live (non-self-hosted). let xmlClosedContentBlock = false if (!isOllamaStream && xmlToolCallText !== null) { const buffered = xmlToolCallText xmlToolCallText = null - const { calls, toolCallRanges } = parseXmlToolCalls( + const { calls, stripToolCallRanges } = parseXmlToolCalls( buffered, allowHy3ToolCalls, ) if (calls.length > 0) { - const stripped = stripRanges(buffered, toolCallRanges).trim() + const stripped = stripRanges(buffered, stripToolCallRanges).trim() const strippedVisible = stripThinkTags(stripped).trim() if (strippedVisible) { // emitTextDelta opens a text block if one is not already open; @@ -2720,6 +3257,39 @@ async function* openaiStreamToAnthropic( ) } + // Truncated streams may end after text deltas without finish_reason. + // Flush any self-hosted buffer so assistant text is not lost at message_stop. + if (isOllamaStream && ollamaTextBuffer) { + throwIfStreamAborted(signal) + if (!hasEmittedContentStart) { + yield { + type: 'content_block_start', + index: contentBlockIndex, + content_block: { type: 'text', text: '' }, + } + hasEmittedContentStart = true + } + throwIfStreamAborted(signal) + yield { + type: 'content_block_delta', + index: contentBlockIndex, + delta: { type: 'text_delta', text: ollamaTextBuffer }, + } + ollamaTextBuffer = '' + if (!hasProcessedFinishReason) { + throwIfStreamAborted(signal) + yield { type: 'content_block_stop', index: contentBlockIndex } + contentBlockIndex++ + hasEmittedContentStart = false + lastStopReason = lastStopReason ?? 'end_turn' + throwIfStreamAborted(signal) + yield { + type: 'message_delta', + delta: { stop_reason: lastStopReason, stop_sequence: null }, + } + } + } + throwIfStreamAborted(signal) yield { type: 'message_stop' } } @@ -2851,11 +3421,15 @@ class OpenAIShimMessages { const promise = (async () => { // A provider override is a complete route, so it must not inherit an - // Azure-style escape hatch intended for the parent route. + // Azure-style escape hatch or parent self-hosted recovery flags intended + // for the active profile. Evaluate recovery against this route only. const requestProcessEnv = self.providerOverride ? { ...process.env, OPENAI_AZURE_STYLE: undefined, + OPENAI_SELF_HOSTED_TOOLS: undefined, + OPENAI_PARSE_TEXT_TOOL_CALLS: undefined, + CLAUDE_CODE_USE_MISTRAL: undefined, } : process.env const request = resolveProviderRequest({ @@ -2864,6 +3438,14 @@ class OpenAIShimMessages { reasoningEffortOverride: self.reasoningEffort, processEnv: requestProcessEnv, }) + // JSON/XML-in-text tool recovery: only when tools are advertised and the + // endpoint is Ollama or other self-hosted compat. Never when tools are + // absent — otherwise benign {"name":...} prose becomes phantom tool_use. + // Use requestProcessEnv so parent profile flags do not leak into overrides. + const enableTextToolCallFallback = + Boolean(params.tools?.length) && + shouldUseSelfHostedToolCompat(request.baseUrl, requestProcessEnv) + const allowedToolNames = toolNamesFromShimParams(params.tools) const response = await self._doRequest(request, params, options, requestProcessEnv) httpResponse = response @@ -2886,7 +3468,14 @@ class OpenAIShimMessages { ? anthropicSsePassthrough(response, request.resolvedModel, streamSignal) : isGeminiStream ? geminiSseToAnthropic(response, request.resolvedModel, streamSignal) - : openaiStreamToAnthropic(response, request.resolvedModel, streamSignal, isLikelyOllamaEndpoint(request.baseUrl), response.url || undefined), + : openaiStreamToAnthropic( + response, + request.resolvedModel, + streamSignal, + enableTextToolCallFallback, + response.url || undefined, + allowedToolNames, + ), options?.signal, cancelBeforeIteration, ) @@ -2921,7 +3510,12 @@ class OpenAIShimMessages { request.resolvedModel, ) } - return self._convertNonStreamingResponse(parsed, request.resolvedModel) + return self._convertNonStreamingResponse( + parsed, + request.resolvedModel, + enableTextToolCallFallback, + allowedToolNames, + ) } } @@ -2946,7 +3540,12 @@ class OpenAIShimMessages { const contentType = response.headers.get('content-type') ?? '' if (contentType.includes('application/json')) { const data = await response.json() - return self._convertNonStreamingResponse(data, request.resolvedModel) + return self._convertNonStreamingResponse( + data, + request.resolvedModel, + enableTextToolCallFallback, + allowedToolNames, + ) } const textBody = await response.text().catch(() => '') @@ -3190,6 +3789,11 @@ class OpenAIShimMessages { request.baseUrl, ), supportsImageInputs: shimConfig.supportsImageInputs, + injectToolResultSemanticBoundary: shouldInjectToolResultSemanticBoundary({ + baseUrl: request.baseUrl, + model: request.resolvedModel, + processEnv: requestProcessEnv, + }), }), ) @@ -4502,8 +5106,13 @@ class OpenAIShimMessages { private _convertNonStreamingResponse( data: NonStreamingOpenAIResponse, model: string, + enableTextToolCallFallback = false, + allowedToolNames?: ReadonlySet, ) { - return convertNonStreamingResponseToAnthropicMessage(data, model) + return convertNonStreamingResponseToAnthropicMessage(data, model, { + enableTextToolCallFallback, + allowedToolNames, + }) } private _convertGeminiToAnthropicResponse( diff --git a/src/services/api/openaiShim.xmlToolCalls.test.ts b/src/services/api/openaiShim.xmlToolCalls.test.ts index fce5495800..ffc8ae9718 100644 --- a/src/services/api/openaiShim.xmlToolCalls.test.ts +++ b/src/services/api/openaiShim.xmlToolCalls.test.ts @@ -48,6 +48,21 @@ function makeChunks(chunks: unknown[]): string[] { return [...chunks.map(c => `data: ${JSON.stringify(c)}\n\n`), 'data: [DONE]\n\n'] } +/** Local mirror of stripRanges for unit assertions (not exported from shim). */ +function stripRangesForTest( + text: string, + ranges: Array<[number, number]>, +): string { + const sorted = [...ranges].sort((a, b) => a[0] - b[0]) + let result = '' + let pos = 0 + for (const [s, e] of sorted) { + result += text.slice(pos, s) + pos = e + } + return result + text.slice(pos) +} + const glmChunk = (content: string, finishReason?: string) => ({ id: 'chatcmpl-glm', object: 'chat.completion.chunk', @@ -234,6 +249,117 @@ describe('parseXmlToolCalls', () => { expect(calls).toHaveLength(1) }) + test('stripToolCallRanges covers every duplicate XML block (1:1 toolCallRanges stays unique)', () => { + // Regression: dedup must not leave a second identical visible. + const block = + 'pwd' + const text = `${block}\n${block}\nnext` + const { calls, toolCallRanges, stripToolCallRanges, stripRangeOwnerIndex } = parseXmlToolCalls(text) + expect(calls).toHaveLength(1) + expect(toolCallRanges).toHaveLength(1) + expect(stripToolCallRanges).toHaveLength(2) + expect(stripRangeOwnerIndex).toHaveLength(2) + // Both strip ranges belong to the same unique call (index 0). + expect(stripRangeOwnerIndex).toEqual([0, 0]) + const stripped = stripRangesForTest(text, stripToolCallRanges) + expect(stripped).not.toContain('') + expect(stripped).toContain('next') + }) + + // Integration test for duplicate XML block stripping with allowlist filtering. + // Tests the full production path: streaming through createOpenAIShimClient with + // tools parameter and verify both duplicate blocks are stripped from text. + test('streaming: duplicate accepted XML blocks stripped from text, tool_use deduped', async () => { + // Mock endpoint that echoes back our streaming response with duplicate XML calls. + const duplicateBlock = + 'echo hello' + const responseText = `${duplicateBlock}\n${duplicateBlock}\nnext` + + const originalBaseUrl = process.env.OPENAI_BASE_URL + const originalApiKey = process.env.OPENAI_API_KEY + const originalFetch = globalThis.fetch as unknown as FetchType | undefined + + process.env.OPENAI_BASE_URL = 'http://127.0.0.1:9999/v1' + process.env.OPENAI_API_KEY = 'test-key' + + let capturedBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + capturedBody = JSON.parse(String(init?.body)) + return new Response( + [ + `data: ${JSON.stringify({ + id: 'chatcmpl-test', + object: 'chat.completion.chunk', + created: 123456789, + model: 'test-model', + choices: [{ index: 0, delta: { content: responseText }, finish_reason: null }], + })}\n\n`, + `data: ${JSON.stringify({ + id: 'chatcmpl-test', + object: 'chat.completion.chunk', + created: 123456789, + model: 'test-model', + choices: [{ index: 0, delta: {}, finish_reason: 'stop' }], + })}\n\n`, + 'data: [DONE]\n\n', + ].join(''), + { + headers: { + 'Content-Type': 'text/event-stream', + 'Transfer-Encoding': 'chunked', + }, + }, + ) + }) as unknown as typeof fetch + + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result: Record[] = [] + + const withResponse = await client.beta.messages.create({ + model: 'test-model', + tools: [{ name: 'Bash', input_schema: {} }], + messages: [ + { role: 'assistant' as const, content: 'Testing duplicates' }, + { role: 'user' as const, content: 'Run bash' }, + ], + max_tokens: 1024, + stream: true, + } as any).withResponse() + + for await (const event of withResponse.data) { + result.push(event) + } + + // Verify we got one tool_use event (duplicates are deduped at call level). + const toolUseEvents = result.filter(e => (e as any).type === 'content_block_start' && (e as any).content_block?.type === 'tool_use') + expect(toolUseEvents).toHaveLength(1) + expect((toolUseEvents[0] as any)?.content_block?.name).toBe('Bash') + + // Find the text_delta event and verify both XML blocks are stripped, "next" preserved. + const textEvents = result.filter(e => (e as any).type === 'content_block_delta' && (e as any).delta?.type === 'text_delta') + const fullText = textEvents.map(e => (e as any).delta.text).join('') + expect(fullText).toContain('next') + expect(fullText).not.toContain('') + + // Teardown: restore original state + if (originalBaseUrl === undefined) { + delete process.env.OPENAI_BASE_URL + } else { + process.env.OPENAI_BASE_URL = originalBaseUrl + } + if (originalApiKey === undefined) { + delete process.env.OPENAI_API_KEY + } else { + process.env.OPENAI_API_KEY = originalApiKey + } + if (originalFetch === undefined) { + delete (globalThis as { fetch?: unknown }).fetch + } else { + globalThis.fetch = originalFetch + } + }) + test('truncated block (no closing tag) still parses', () => { const text = 'ls' const { calls } = parseXmlToolCalls(text) @@ -506,3 +632,257 @@ describe('GLM streaming — XML tool calls', () => { }) }) }) + +// Self-hosted buffering path (LAN llama-server) enables isOllamaStream-style +// text buffering. XML recovery must still run at finish — otherwise Qwen/GLM +// XML tool calls are flushed as plain text with end_turn. Also cover +// finish_reason tool_calls without structured delta.tool_calls. +describe('Self-hosted streaming — XML tool calls with tools advertised', () => { + let originalFetch: FetchType + let originalOpenAIApiKey: string | undefined + let originalOpenAIBaseUrl: string | undefined + beforeEach(() => { + originalFetch = globalThis.fetch + originalOpenAIApiKey = process.env.OPENAI_API_KEY + originalOpenAIBaseUrl = process.env.OPENAI_BASE_URL + process.env.OPENAI_API_KEY = 'none' + process.env.OPENAI_BASE_URL = 'http://192.168.1.10:8080/v1' + }) + afterEach(() => { + globalThis.fetch = originalFetch + if (originalOpenAIApiKey === undefined) { + delete process.env.OPENAI_API_KEY + } else { + process.env.OPENAI_API_KEY = originalOpenAIApiKey + } + if (originalOpenAIBaseUrl === undefined) { + delete process.env.OPENAI_BASE_URL + } else { + process.env.OPENAI_BASE_URL = originalOpenAIBaseUrl + } + }) + + test('recovers Qwen/GLM XML tool_use from buffered self-hosted stream', async () => { + const chunk = (content: string, finishReason?: string) => ({ + id: 'chatcmpl-llama', + object: 'chat.completion.chunk', + model: 'qwen3.6:35b', + choices: [ + { + index: 0, + delta: { content }, + finish_reason: finishReason ?? null, + }, + ], + }) + + globalThis.fetch = (async () => + makeSseResponse( + makeChunks([ + chunk( + 'pwd', + ), + chunk('', 'stop'), + ]), + )) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result = await client.beta.messages + .create({ + model: 'qwen3.6:35b', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [ + { + name: 'Bash', + description: 'run shell', + input_schema: { type: 'object', properties: {} }, + }, + ], + max_tokens: 64, + stream: true, + }) + .withResponse() + + const events: Record[] = [] + for await (const event of result.data) events.push(event) + + const starts = events.filter( + e => + e.type === 'content_block_start' && + (e.content_block as Record)?.type === 'tool_use', + ) + expect(starts).toHaveLength(1) + expect((starts[0].content_block as Record).name).toBe('Bash') + + const text = events + .filter( + e => + e.type === 'content_block_delta' && + (e.delta as Record)?.type === 'text_delta', + ) + .map(e => (e.delta as Record).text) + .join('') + expect(text).not.toContain('') + }) + + test('recovers JSON text tools when finish_reason is tool_calls without delta.tool_calls', async () => { + const chunk = (content: string, finishReason?: string) => ({ + id: 'chatcmpl-llama', + object: 'chat.completion.chunk', + model: 'qwen3.6:35b', + choices: [ + { + index: 0, + delta: { content }, + finish_reason: finishReason ?? null, + }, + ], + }) + + globalThis.fetch = (async () => + makeSseResponse( + makeChunks([ + chunk('{"name":"Bash","arguments":{"command":"pwd"}}'), + chunk('', 'tool_calls'), + ]), + )) as unknown as FetchType + + const client = createOpenAIShimClient({}) as OpenAIShimClient + const result = await client.beta.messages + .create({ + model: 'qwen3.6:35b', + messages: [{ role: 'user', content: 'run pwd' }], + tools: [ + { + name: 'Bash', + description: 'run shell', + input_schema: { type: 'object', properties: {} }, + }, + ], + max_tokens: 64, + stream: true, + }) + .withResponse() + + const events: Record[] = [] + for await (const event of result.data) events.push(event) + + const starts = events.filter( + e => + e.type === 'content_block_start' && + (e.content_block as Record)?.type === 'tool_use', + ) + expect(starts).toHaveLength(1) + expect((starts[0].content_block as Record).name).toBe('Bash') + + const messageDelta = events.find(e => e.type === 'message_delta') as + | Record + | undefined + expect( + (messageDelta?.delta as Record | undefined)?.stop_reason, + ).toBe('tool_use') + }) +}) + +describe('providerOverride does not inherit parent self-hosted recovery flags', () => { + let originalFetch: FetchType + let originalOpenAIApiKey: string | undefined + let originalOpenAIBaseUrl: string | undefined + let originalSelfHosted: string | undefined + beforeEach(() => { + originalFetch = globalThis.fetch + originalOpenAIApiKey = process.env.OPENAI_API_KEY + originalOpenAIBaseUrl = process.env.OPENAI_BASE_URL + originalSelfHosted = process.env.OPENAI_SELF_HOSTED_TOOLS + process.env.OPENAI_API_KEY = 'parent-key' + // Parent profile is self-hosted with recovery enabled. + process.env.OPENAI_BASE_URL = 'https://llama.example.com:8443/v1' + process.env.OPENAI_SELF_HOSTED_TOOLS = '1' + }) + afterEach(() => { + globalThis.fetch = originalFetch + if (originalOpenAIApiKey === undefined) { + delete process.env.OPENAI_API_KEY + } else { + process.env.OPENAI_API_KEY = originalOpenAIApiKey + } + if (originalOpenAIBaseUrl === undefined) { + delete process.env.OPENAI_BASE_URL + } else { + process.env.OPENAI_BASE_URL = originalOpenAIBaseUrl + } + if (originalSelfHosted === undefined) { + delete process.env.OPENAI_SELF_HOSTED_TOOLS + } else { + process.env.OPENAI_SELF_HOSTED_TOOLS = originalSelfHosted + } + }) + + test('remote override leaves tool-shaped text as text (not tool_use)', async () => { + const chunk = (content: string, finishReason?: string) => ({ + id: 'chatcmpl-remote', + object: 'chat.completion.chunk', + model: 'gpt-4o', + choices: [ + { + index: 0, + delta: { content }, + finish_reason: finishReason ?? null, + }, + ], + }) + + globalThis.fetch = (async () => + makeSseResponse( + makeChunks([ + chunk( + 'Here is an example: {"name":"Bash","arguments":{"command":"ls"}}', + ), + chunk('', 'stop'), + ]), + )) as unknown as FetchType + + const client = createOpenAIShimClient({ + providerOverride: { + model: 'gpt-4o', + baseURL: 'https://api.openai.com/v1', + apiKey: 'sk-override', + }, + }) as OpenAIShimClient + const result = await client.beta.messages + .create({ + model: 'gpt-4o', + messages: [{ role: 'user', content: 'show an example' }], + tools: [ + { + name: 'Bash', + description: 'run shell', + input_schema: { type: 'object', properties: {} }, + }, + ], + max_tokens: 64, + stream: true, + }) + .withResponse() + + const events: Record[] = [] + for await (const event of result.data) events.push(event) + + const toolStarts = events.filter( + e => + e.type === 'content_block_start' && + (e.content_block as Record)?.type === 'tool_use', + ) + expect(toolStarts).toHaveLength(0) + + const text = events + .filter( + e => + e.type === 'content_block_delta' && + (e.delta as Record)?.type === 'text_delta', + ) + .map(e => (e.delta as Record).text) + .join('') + expect(text).toContain('{"name":"Bash"') + }) +}) diff --git a/src/services/api/providerConfig.local.test.ts b/src/services/api/providerConfig.local.test.ts index 2849dcfa9d..9562c2c087 100644 --- a/src/services/api/providerConfig.local.test.ts +++ b/src/services/api/providerConfig.local.test.ts @@ -9,6 +9,9 @@ import { modelRequiresResponsesApi, resolveProviderRequest, shouldAttemptLocalToollessRetry, + shouldInjectToolResultSemanticBoundary, + shouldUseSelfHostedToolCompat, + TOOL_RESULT_SEMANTIC_PLACEHOLDER, } from './providerConfig.js' const originalEnv = { @@ -90,6 +93,120 @@ test('treats public hosts as remote', () => { expect(isLocalProviderUrl('http://[2001:4860:4860::8888]:11434/v1')).toBe(false) }) +test('semantic tool-result boundary is Mistral-only', () => { + expect(TOOL_RESULT_SEMANTIC_PLACEHOLDER).toBe('[Tool results received]') + + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'http://localhost:8080/v1', + model: 'qwen3.6:35b', + processEnv: {}, + }), + ).toBe(false) + + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'https://api.openai.com/v1', + model: 'gpt-4o', + processEnv: {}, + }), + ).toBe(false) + + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'https://api.mistral.ai/v1', + model: 'mistral-large-latest', + processEnv: {}, + }), + ).toBe(true) + + // Substring must not match — only hostname mistral.ai / *.mistral.ai + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'https://mistral.ai-proxy.example/v1', + model: 'qwen3.6:35b', + processEnv: {}, + }), + ).toBe(false) + + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'http://localhost:8080/v1', + model: 'devstral-small', + processEnv: {}, + }), + ).toBe(true) + + // Codestral is Mistral-class even behind a non-mistral.ai proxy hostname. + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'https://llm-proxy.example.com/v1', + model: 'codestral-latest', + processEnv: {}, + }), + ).toBe(true) + + expect( + shouldInjectToolResultSemanticBoundary({ + baseUrl: 'http://localhost:8080/v1', + model: 'qwen3', + processEnv: { CLAUDE_CODE_USE_MISTRAL: '1' }, + }), + ).toBe(true) +}) + +test('self-hosted tool compat is env- or local/Ollama-gated (any host when env set)', () => { + expect( + shouldUseSelfHostedToolCompat('https://llama.example.com:8443/v1', {}), + ).toBe(false) + + expect( + shouldUseSelfHostedToolCompat('https://llama.example.com:8443/v1', { + OPENAI_SELF_HOSTED_TOOLS: '1', + }), + ).toBe(true) + + expect( + shouldUseSelfHostedToolCompat('https://proxy.example.com/v1', { + OPENAI_PARSE_TEXT_TOOL_CALLS: '1', + }), + ).toBe(true) + + // llama-server on arbitrary local port + expect( + shouldUseSelfHostedToolCompat('http://127.0.0.1:8080/v1', {}), + ).toBe(true) + expect( + shouldUseSelfHostedToolCompat('http://192.168.1.50:9000/v1', {}), + ).toBe(true) + + expect( + shouldUseSelfHostedToolCompat('http://gpu.local:8080/v1', {}), + ).toBe(true) + + expect( + shouldUseSelfHostedToolCompat('http://remote.example.com:11434/v1', {}), + ).toBe(true) + + // Explicit Disabled (OPENAI_SELF_HOSTED_TOOLS=0) must force recovery off even + // for loopback / RFC1918 / .local URLs. + expect( + shouldUseSelfHostedToolCompat('http://127.0.0.1:8080/v1', { + OPENAI_SELF_HOSTED_TOOLS: '0', + }), + ).toBe(false) + expect( + shouldUseSelfHostedToolCompat('http://192.168.1.50:9000/v1', { + OPENAI_SELF_HOSTED_TOOLS: 'false', + }), + ).toBe(false) + expect( + shouldUseSelfHostedToolCompat('http://gpu.local:8080/v1', { + OPENAI_PARSE_TEXT_TOOL_CALLS: 'off', + }), + ).toBe(false) +}) + test('creates a cache scope for local openai-compatible providers', () => { process.env.CLAUDE_CODE_USE_OPENAI = '1' process.env.OPENAI_BASE_URL = 'http://localhost:1234/v1' diff --git a/src/services/api/providerConfig.ts b/src/services/api/providerConfig.ts index dc023a3d4f..ea32b84010 100644 --- a/src/services/api/providerConfig.ts +++ b/src/services/api/providerConfig.ts @@ -10,7 +10,7 @@ import { type CodexCredentialBlob, } from '../../utils/codexCredentials.js' import { logForDebugging } from '../../utils/debug.js' -import { isEnvTruthy } from '../../utils/envUtils.js' +import { isEnvDefinedFalsy, isEnvTruthy } from '../../utils/envUtils.js' import { asTrimmedString, parseChatgptAccountId, @@ -675,6 +675,109 @@ export function isLikelyOllamaEndpoint(baseUrl: string | undefined): boolean { } } +/** + * Synthetic assistant text the OpenAI shim injects between `tool` and `user` + * for Mistral/Devstral Jinja role sequencing only. Never use this as a + * continuation signal for llama-server / Ollama / vLLM — those backends treat + * it as real assistant content and often echo it, stalling the tool loop. + */ +export const TOOL_RESULT_SEMANTIC_PLACEHOLDER = '[Tool results received]' + +/** + * Opt-in for self-hosted OpenAI-compatible tool behaviour (llama-server, vLLM, + * LM Studio, remote tunnels). When set, host detection is not required — any + * base URL (public domain, reverse proxy, non-standard port) is treated as a + * text-tool-compat backend. + * + * Also accepted: OPENAI_PARSE_TEXT_TOOL_CALLS (historical alias). + */ +export const OPENAI_SELF_HOSTED_TOOLS_ENV = 'OPENAI_SELF_HOSTED_TOOLS' + +/** + * Whether to inject the Mistral-only tool→assistant→user semantic boundary. + * Other OpenAI-compatible backends (llama-server + Qwen, Ollama, vLLM, OpenAI) + * must keep tool→user (or tool as last message) so Jinja templates stay valid. + */ +export function shouldInjectToolResultSemanticBoundary(options?: { + baseUrl?: string + model?: string + processEnv?: NodeJS.ProcessEnv +}): boolean { + const processEnv = options?.processEnv ?? process.env + if (isEnvTruthy(processEnv.CLAUDE_CODE_USE_MISTRAL)) { + return true + } + + const baseUrl = + options?.baseUrl ?? + processEnv.MISTRAL_BASE_URL ?? + processEnv.OPENAI_BASE_URL ?? + '' + // Hostname only — avoid false positives like https://mistral.ai-proxy.example/v1 + try { + const host = new URL(baseUrl).hostname.toLowerCase() + if ( + host === 'mistral.ai' || + host === 'api.mistral.ai' || + host.endsWith('.mistral.ai') + ) { + return true + } + } catch { + // Invalid URL — fall through to model-name detection. + } + + const model = ( + options?.model ?? + processEnv.MISTRAL_MODEL ?? + processEnv.OPENAI_MODEL ?? + '' + ).toLowerCase() + return /\b(devstral|mistral|ministral|codestral)\b/.test(model) +} + +/** + * Self-hosted tool-call recovery (JSON-in-text + stream buffering). + * + * True when: + * - OPENAI_SELF_HOSTED_TOOLS=1 (or OPENAI_PARSE_TEXT_TOOL_CALLS=1) — any host + * - base URL is loopback / RFC1918 / .local (llama-server on LAN, any port) + * - endpoint looks like Ollama + * + * Explicit `0`/`false`/`off` on either flag forces recovery off (including + * local URLs) so a profile/UI "Disabled" selection is honoured. + * + * Host detection is intentionally not required when the env flag is set so a + * publicly reverse-proxied llama-server works the same as localhost:8080. + * + * Callers with a providerOverride must pass that route's processEnv (with + * parent self-hosted flags cleared) so recovery is not inherited from the + * parent profile. + */ +export function shouldUseSelfHostedToolCompat( + baseUrl?: string, + processEnv: NodeJS.ProcessEnv = process.env, +): boolean { + // Explicit disable wins over local/Ollama auto-detect and over a conflicting + // truthy alias — profile "Disabled" sets OPENAI_SELF_HOSTED_TOOLS=0. + if ( + isEnvDefinedFalsy(processEnv[OPENAI_SELF_HOSTED_TOOLS_ENV]) || + isEnvDefinedFalsy(processEnv.OPENAI_PARSE_TEXT_TOOL_CALLS) + ) { + return false + } + if (isEnvTruthy(processEnv[OPENAI_SELF_HOSTED_TOOLS_ENV])) { + return true + } + if (isEnvTruthy(processEnv.OPENAI_PARSE_TEXT_TOOL_CALLS)) { + return true + } + if (isLocalProviderUrl(baseUrl)) { + return true + } + return isLikelyOllamaEndpoint(baseUrl) +} + export function isDirectLocalOllamaEndpoint(baseUrl: string | undefined): boolean { if (!baseUrl) return false try { diff --git a/src/utils/config.ts b/src/utils/config.ts index e922754771..b1c4dc9aae 100644 --- a/src/utils/config.ts +++ b/src/utils/config.ts @@ -237,6 +237,13 @@ export type ProviderProfile = { * Applied to OpenAI-compatible providers when resolving runtime limits. */ maxContextLength?: number + /** + * Per-profile self-hosted tool compatibility (llama-server, vLLM, Ollama, + * etc.). When this profile is active, OpenClaude enables JSON-in-text + * tool recovery and related self-hosted tool behaviour for this provider + * only — no shell env required. Other profiles are unaffected. + */ + selfHostedTools?: boolean } export type GlobalConfig = { diff --git a/src/utils/providerProfile.test.ts b/src/utils/providerProfile.test.ts index 0e162bcad6..9523e1486a 100644 --- a/src/utils/providerProfile.test.ts +++ b/src/utils/providerProfile.test.ts @@ -2654,6 +2654,43 @@ test('openai launch preserves invalid live pooled credentials for launch validat assert.equal(env.OPENAI_API_KEY, undefined) assert.equal(hasInvalidOpenAICredentialPool(env.OPENAI_API_KEYS), true) }) + +test('openai launch preserves persisted self-hosted tools flags on relaunch', async () => { + const env = await buildLaunchEnv({ + profile: 'openai', + persisted: profile('openai', { + OPENAI_BASE_URL: 'http://172.16.30.12:8081/v1', + OPENAI_MODEL: 'qwen3.6:35b', + OPENAI_SELF_HOSTED_TOOLS: '1', + OPENAI_PARSE_TEXT_TOOL_CALLS: '1', + }), + goal: 'balanced', + processEnv: {}, + }) + + assert.equal(env.OPENAI_SELF_HOSTED_TOOLS, '1') + assert.equal(env.OPENAI_PARSE_TEXT_TOOL_CALLS, '1') + assert.equal(env.OPENAI_BASE_URL, 'http://172.16.30.12:8081/v1') +}) + +test('openai launch prefers shell self-hosted tools flags over persisted', async () => { + const env = await buildLaunchEnv({ + profile: 'openai', + persisted: profile('openai', { + OPENAI_BASE_URL: 'http://172.16.30.12:8081/v1', + OPENAI_MODEL: 'qwen3.6:35b', + OPENAI_SELF_HOSTED_TOOLS: '1', + }), + goal: 'balanced', + processEnv: { + OPENAI_SELF_HOSTED_TOOLS: 'true', + OPENAI_PARSE_TEXT_TOOL_CALLS: '1', + }, + }) + + assert.equal(env.OPENAI_SELF_HOSTED_TOOLS, 'true') + assert.equal(env.OPENAI_PARSE_TEXT_TOOL_CALLS, '1') +}) test('openai launch lets a live singular key override a saved pool', async () => { const env = await buildLaunchEnv({ profile: 'openai', diff --git a/src/utils/providerProfile.ts b/src/utils/providerProfile.ts index 64cbaf1919..6ad3b6a4e0 100644 --- a/src/utils/providerProfile.ts +++ b/src/utils/providerProfile.ts @@ -80,6 +80,11 @@ const PROFILE_ENV_KEYS = [ 'OPENAI_AUTH_HEADER_VALUE', 'OPENAI_API_KEYS', 'OPENAI_API_KEY', + 'OPENAI_SELF_HOSTED_TOOLS', + 'OPENAI_PARSE_TEXT_TOOL_CALLS', + // Provenance: flag was applied from a persisted startup profile, not shell. + 'OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS', + 'OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS', 'GITHUB_COPILOT_KEY', 'GITHUB_ENTERPRISE_URL', 'CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS', @@ -167,6 +172,14 @@ export type ProfileEnv = { OPENAI_AUTH_HEADER_VALUE?: string OPENAI_API_KEYS?: string OPENAI_API_KEY?: string + /** Set from active profile.selfHostedTools — not meant for manual shell use. */ + OPENAI_SELF_HOSTED_TOOLS?: string + /** Historical alias still cleared/applied with profile env swaps. */ + OPENAI_PARSE_TEXT_TOOL_CALLS?: string + /** Set when OPENAI_SELF_HOSTED_TOOLS came from a persisted startup profile. */ + OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS?: string + /** Set when OPENAI_PARSE_TEXT_TOOL_CALLS came from a persisted startup profile. */ + OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS?: string GITHUB_COPILOT_KEY?: string GITHUB_ENTERPRISE_URL?: string CODEX_API_KEY?: string @@ -2121,6 +2134,49 @@ export async function buildLaunchEnv(options: { env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = contextWindows } + // Self-hosted tool recovery flags (profile.selfHostedTools / shell override). + // Shell wins over persisted, same as API format / context windows. Without + // this, applyProfileEnvToProcessEnv clears PROFILE_ENV_KEYS on relaunch and + // drops OPENAI_SELF_HOSTED_TOOLS even when the legacy profile file had it. + // When the value comes only from the persisted startup profile, mark + // provenance so later profile activation does not treat it as a shell export. + // Prefer shell whenever it is defined (including empty string); only fall + // back to persisted when shell is undefined. + const shellSelfHostedToolsFlag = processEnv.OPENAI_SELF_HOSTED_TOOLS + const persistedSelfHostedToolsFlag = usePersistedOpenAIConfig + ? persistedEnv.OPENAI_SELF_HOSTED_TOOLS + : undefined + const selfHostedToolsFlag = + shellSelfHostedToolsFlag !== undefined + ? shellSelfHostedToolsFlag + : persistedSelfHostedToolsFlag + if (selfHostedToolsFlag !== undefined) { + env.OPENAI_SELF_HOSTED_TOOLS = selfHostedToolsFlag + if ( + shellSelfHostedToolsFlag === undefined && + persistedSelfHostedToolsFlag !== undefined + ) { + env.OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS = '1' + } + } + const shellParseTextToolCallsFlag = processEnv.OPENAI_PARSE_TEXT_TOOL_CALLS + const persistedParseTextToolCallsFlag = usePersistedOpenAIConfig + ? persistedEnv.OPENAI_PARSE_TEXT_TOOL_CALLS + : undefined + const parseTextToolCallsFlag = + shellParseTextToolCallsFlag !== undefined + ? shellParseTextToolCallsFlag + : persistedParseTextToolCallsFlag + if (parseTextToolCallsFlag !== undefined) { + env.OPENAI_PARSE_TEXT_TOOL_CALLS = parseTextToolCallsFlag + if ( + shellParseTextToolCallsFlag === undefined && + persistedParseTextToolCallsFlag !== undefined + ) { + env.OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS = '1' + } + } + return buildCompatibilityProcessEnv({ processEnv, compatibilityMode: 'openai', diff --git a/src/utils/providerProfiles.test.ts b/src/utils/providerProfiles.test.ts index 77c69959cd..d8d24c79a0 100644 --- a/src/utils/providerProfiles.test.ts +++ b/src/utils/providerProfiles.test.ts @@ -37,6 +37,10 @@ const RESTORED_KEYS = [ 'OPENAI_AUTH_HEADER_VALUE', 'OPENAI_API_KEYS', 'OPENAI_API_KEY', + 'OPENAI_SELF_HOSTED_TOOLS', + 'OPENAI_PARSE_TEXT_TOOL_CALLS', + 'OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS', + 'OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS', 'GITHUB_COPILOT_KEY', 'GITHUB_ENTERPRISE_URL', 'CODEX_API_KEY', @@ -325,6 +329,197 @@ describe('applyProviderProfileToProcessEnv', () => { expect(getFreshAPIProvider()).toBe('openai') }, 20_000) + test('selfHostedTools on profile sets OPENAI_SELF_HOSTED_TOOLS only while active', async () => { + const { + applyProviderProfileToProcessEnv, + clearProviderProfileEnvFromProcessEnv, + _resetShellSelfHostedOverridesForTests, + } = await importFreshProviderProfileModules() + _resetShellSelfHostedOverridesForTests() + + applyProviderProfileToProcessEnv( + buildProfile({ + selfHostedTools: true, + baseUrl: 'http://172.16.30.12:8081/v1', + model: 'qwen3.6:35b', + }), + ) + expect(process.env.OPENAI_SELF_HOSTED_TOOLS).toBe('1') + expect(process.env.OPENAI_BASE_URL).toBe('http://172.16.30.12:8081/v1') + + clearProviderProfileEnvFromProcessEnv() + expect(process.env.OPENAI_SELF_HOSTED_TOOLS).toBeUndefined() + + applyProviderProfileToProcessEnv( + buildProfile({ + selfHostedTools: false, + baseUrl: 'https://api.openai.com/v1', + model: 'gpt-4o', + }), + ) + // Explicit Disabled persists as '0' so local auto-detect cannot re-enable. + // String() avoids TS control-flow narrowing after delete above. + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('0') + }) + + test('shell OPENAI_SELF_HOSTED_TOOLS survives profile activation without UI flag', async () => { + const { + applyProviderProfileToProcessEnv, + _resetShellSelfHostedOverridesForTests, + } = await importFreshProviderProfileModules() + _resetShellSelfHostedOverridesForTests() + + process.env.OPENAI_SELF_HOSTED_TOOLS = '1' + process.env.OPENAI_PARSE_TEXT_TOOL_CALLS = '1' + + applyProviderProfileToProcessEnv( + buildProfile({ + selfHostedTools: false, + baseUrl: 'https://llama.example.com:8443/v1', + model: 'qwen3.6:35b', + }), + ) + + expect(process.env.OPENAI_SELF_HOSTED_TOOLS).toBe('1') + expect(process.env.OPENAI_PARSE_TEXT_TOOL_CALLS).toBe('1') + expect(process.env.OPENAI_BASE_URL).toBe('https://llama.example.com:8443/v1') + }) + + test('shell OPENAI_SELF_HOSTED_TOOLS override keeps process env aligned with profile', async () => { + const { + applyProviderProfileToProcessEnv, + applyActiveProviderProfileFromConfig, + _resetShellSelfHostedOverridesForTests, + } = await importFreshProviderProfileModules() + _resetShellSelfHostedOverridesForTests() + + // Shell forces recovery on while profile UI says Disabled — apply still + // restores shell, and alignment must treat that as expected (no re-loop). + process.env.OPENAI_SELF_HOSTED_TOOLS = '1' + + const profile = buildProfile({ + id: 'provider_self_hosted_shell_align', + selfHostedTools: false, + baseUrl: 'https://llama.example.com:8443/v1', + model: 'qwen3.6:35b', + }) + mockConfigState = { + ...mockConfigState, + providerProfiles: [profile], + activeProviderProfileId: profile.id, + } + + applyProviderProfileToProcessEnv(profile) + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('1') + + const restored = applyActiveProviderProfileFromConfig( + { ...mockConfigState } as any, + { processEnv: process.env, force: false }, + ) + // Aligned → no forced re-apply needed; profile still active, shell flag kept. + expect(restored?.id).toBe(profile.id) + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('1') + }) + + test('isProcessEnvAlignedWithProfile detects OPENAI_SELF_HOSTED_TOOLS drift', async () => { + const { + applyProviderProfileToProcessEnv, + applyActiveProviderProfileFromConfig, + _resetShellSelfHostedOverridesForTests, + } = await importFreshProviderProfileModules() + _resetShellSelfHostedOverridesForTests() + + const profile = buildProfile({ + id: 'provider_self_hosted_align', + selfHostedTools: true, + baseUrl: 'https://llama.example.com:8443/v1', + model: 'qwen3.6:35b', + }) + + mockConfigState = { + ...mockConfigState, + providerProfiles: [profile], + activeProviderProfileId: profile.id, + } + + applyProviderProfileToProcessEnv(profile) + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('1') + + // Simulate external clear of the flag while profile is still marked applied. + delete process.env.OPENAI_SELF_HOSTED_TOOLS + + // Alignment must fail → re-apply restores the profile flag. + const restored = applyActiveProviderProfileFromConfig( + { + ...mockConfigState, + } as any, + { processEnv: process.env, force: false }, + ) + expect(restored?.id).toBe(profile.id) + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('1') + }) + + test('selfHostedTools true→false activation clears flag (prior profile not shell)', async () => { + const { + applyProviderProfileToProcessEnv, + _resetShellSelfHostedOverridesForTests, + } = await importFreshProviderProfileModules() + _resetShellSelfHostedOverridesForTests() + + // No shell export — first activation enables via profile. + delete process.env.OPENAI_SELF_HOSTED_TOOLS + delete process.env.OPENAI_PARSE_TEXT_TOOL_CALLS + + applyProviderProfileToProcessEnv( + buildProfile({ + id: 'provider_self_hosted_on', + selfHostedTools: true, + baseUrl: 'https://llama.example.com:8443/v1', + model: 'qwen3.6:35b', + }), + ) + // String() avoids TS control-flow narrowing after delete above. + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('1') + + // Second activation with selfHostedTools=false must force-disable — must not + // treat the previous profile's '1' as a shell override. + applyProviderProfileToProcessEnv( + buildProfile({ + id: 'provider_self_hosted_off', + selfHostedTools: false, + baseUrl: 'https://llama.example.com:8443/v1', + model: 'qwen3.6:35b', + }), + ) + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('0') + }) + + test('shell OPENAI_SELF_HOSTED_TOOLS survives clearActiveProviderProfile', async () => { + const { + applyProviderProfileToProcessEnv, + clearActiveProviderProfile, + _resetShellSelfHostedOverridesForTests, + } = await importFreshProviderProfileModules() + _resetShellSelfHostedOverridesForTests() + + process.env.OPENAI_SELF_HOSTED_TOOLS = '1' + process.env.OPENAI_PARSE_TEXT_TOOL_CALLS = '1' + + applyProviderProfileToProcessEnv( + buildProfile({ + selfHostedTools: true, + baseUrl: 'https://llama.example.com:8443/v1', + model: 'qwen3.6:35b', + }), + ) + expect(String(process.env.OPENAI_SELF_HOSTED_TOOLS)).toBe('1') + + // Switching to built-in Anthropic must not delete genuine shell overrides. + clearActiveProviderProfile() + expect(process.env.OPENAI_SELF_HOSTED_TOOLS).toBe('1') + expect(process.env.OPENAI_PARSE_TEXT_TOOL_CALLS).toBe('1') + }) + test('mistral profile sets CLAUDE_CODE_USE_MISTRAL and clears openai flags', async () => { const { applyProviderProfileToProcessEnv } = await importFreshProviderProfileModules() diff --git a/src/utils/providerProfiles.ts b/src/utils/providerProfiles.ts index f9810d6721..9585441e18 100644 --- a/src/utils/providerProfiles.ts +++ b/src/utils/providerProfiles.ts @@ -68,6 +68,48 @@ import { getSettings_DEPRECATED } from './settings/settings.js' export type { ProviderPreset } from '../integrations/index.js' +const PROFILE_ENV_APPLIED_FLAG = 'CLAUDE_CODE_PROVIDER_PROFILE_ENV_APPLIED' +const PROFILE_ENV_APPLIED_ID = 'CLAUDE_CODE_PROVIDER_PROFILE_ENV_APPLIED_ID' + +/** + * Shell-origin overrides for self-hosted tool flags. Captured once from the + * pre-managed environment (before PROFILE_ENV_APPLIED) so a prior profile's + * OPENAI_SELF_HOSTED_TOOLS=1 is never mistaken for a shell export on the next + * activation. Direct selfHostedTools true→false must apply false. + */ +let shellSelfHostedToolsOverride: string | undefined +let shellParseTextToolCallsOverride: string | undefined +let shellSelfHostedOverridesCaptured = false + +function captureShellSelfHostedOverridesIfNeeded(): void { + // Only sample process.env when no managed profile has been applied yet. + if (shellSelfHostedOverridesCaptured) { + return + } + if (process.env[PROFILE_ENV_APPLIED_FLAG] === '1') { + // Session already managed — do not treat current values as shell. + shellSelfHostedOverridesCaptured = true + return + } + // Startup may have applied OPENAI_SELF_HOSTED_TOOLS from the persisted + // profile file without PROFILE_ENV_APPLIED. Those values are marked with + // OPENCLAUDE_STARTUP_* and must not be captured as shell overrides. + if (process.env.OPENCLAUDE_STARTUP_SELF_HOSTED_TOOLS !== '1') { + shellSelfHostedToolsOverride = process.env.OPENAI_SELF_HOSTED_TOOLS + } + if (process.env.OPENCLAUDE_STARTUP_PARSE_TEXT_TOOL_CALLS !== '1') { + shellParseTextToolCallsOverride = process.env.OPENAI_PARSE_TEXT_TOOL_CALLS + } + shellSelfHostedOverridesCaptured = true +} + +/** Test helper: reset shell-override provenance between tests. */ +export function _resetShellSelfHostedOverridesForTests(): void { + shellSelfHostedToolsOverride = undefined + shellParseTextToolCallsOverride = undefined + shellSelfHostedOverridesCaptured = false +} + export type ProviderProfileInput = { provider?: ProviderProfile['provider'] name: string @@ -81,15 +123,60 @@ export type ProviderProfileInput = { authHeaderValue?: ProviderProfile['authHeaderValue'] customHeaders?: ProviderProfile['customHeaders'] maxContextLength?: ProviderProfile['maxContextLength'] + selfHostedTools?: boolean } export type ProviderPresetDefaults = Omit & { provider: ProviderProfile['provider'] requiresApiKey: boolean + selfHostedTools?: boolean } -const PROFILE_ENV_APPLIED_FLAG = 'CLAUDE_CODE_PROVIDER_PROFILE_ENV_APPLIED' -const PROFILE_ENV_APPLIED_ID = 'CLAUDE_CODE_PROVIDER_PROFILE_ENV_APPLIED_ID' +/** OpenAI-compatible profiles can toggle per-profile self-hosted tool compat. */ +export function providerProfileSupportsSelfHostedTools( + provider: string, +): boolean { + return resolveProfileCompatibility(provider).compatibilityMode === 'openai' +} + +/** + * Profile UI toggle → managed env value. + * - true → '1' (force recovery on any host) + * - false → '0' (force recovery off, including local URLs) + * - undefined → unset (local/Ollama auto-detect still applies) + */ +function selfHostedToolsEnvValue( + selfHostedTools?: boolean, +): '1' | '0' | undefined { + if (selfHostedTools === true) return '1' + if (selfHostedTools === false) return '0' + return undefined +} + +/** Attach OPENAI_SELF_HOSTED_TOOLS from the profile UI toggle. */ +export function applySelfHostedToolsProfileEnv( + env: ProfileEnv, + selfHostedTools?: boolean, +): ProfileEnv { + const value = selfHostedToolsEnvValue(selfHostedTools) + if (value === undefined) { + return env + } + return { ...env, OPENAI_SELF_HOSTED_TOOLS: value } +} + +/** Re-apply shell-origin self-hosted flags after managed profile env cleanup. */ +export function restoreShellSelfHostedOverrides( + processEnv: NodeJS.ProcessEnv = process.env, +): void { + // Use !== undefined so an explicit empty shell value still wins. + if (shellSelfHostedToolsOverride !== undefined) { + processEnv.OPENAI_SELF_HOSTED_TOOLS = shellSelfHostedToolsOverride + } + if (shellParseTextToolCallsOverride !== undefined) { + processEnv.OPENAI_PARSE_TEXT_TOOL_CALLS = shellParseTextToolCallsOverride + } +} type ProfileCompatibilityMode = | 'anthropic' @@ -374,6 +461,14 @@ function sanitizeProfile(profile: ProviderProfile): ProviderProfile | null { if (maxContextLength !== undefined) { sanitized.maxContextLength = maxContextLength } + if ( + providerProfileSupportsSelfHostedTools(provider) && + typeof profile.selfHostedTools === 'boolean' + ) { + // Persist both true and false so UI "Disabled" survives reload and can + // force recovery off for local URLs (auto-detect otherwise re-enables it). + sanitized.selfHostedTools = profile.selfHostedTools + } return sanitized } @@ -415,6 +510,7 @@ function toProfile( authHeaderValue: input.authHeaderValue, customHeaders: input.customHeaders, maxContextLength: input.maxContextLength, + selfHostedTools: input.selfHostedTools, }) } @@ -506,6 +602,8 @@ export function getProviderPresetDefaults( ? process.env.ANTHROPIC_AUTH_TOKEN?.trim() || undefined : metadata.apiKey, requiresApiKey: metadata.requiresApiKey, + // Ollama is always self-hosted; enable tool-text recovery by default. + selfHostedTools: preset === 'ollama' ? true : undefined, } } @@ -770,6 +868,13 @@ function isProcessEnvAlignedWithProfile( processEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS, expectedContextWindows, ) && + sameOptionalEnvValue( + processEnv.OPENAI_SELF_HOSTED_TOOLS, + // Match applyProviderProfileToProcessEnv: shell override wins when present. + shellSelfHostedToolsOverride !== undefined + ? shellSelfHostedToolsOverride + : selfHostedToolsEnvValue(profile.selfHostedTools), + ) && (!includeApiKey || sameOptionalEnvValue(processEnv.OPENAI_API_KEY, profile.apiKey)) && (profile.baseUrl?.toLowerCase().includes('bankr') @@ -873,7 +978,16 @@ export function clearActiveProviderProfile( export function clearProviderProfileEnvFromProcessEnv( processEnv: NodeJS.ProcessEnv = process.env, ): void { + // Capture shell-origin values before managed cleanup deletes PROFILE_ENV_KEYS + // (includes OPENAI_SELF_HOSTED_TOOLS). clearActiveProviderProfile / Anthropic + // selection use this path and do not re-run applyProviderProfileToProcessEnv. + if (processEnv === process.env) { + captureShellSelfHostedOverridesIfNeeded() + } clearManagedProfileEnv(processEnv) + if (processEnv === process.env) { + restoreShellSelfHostedOverrides(processEnv) + } delete processEnv[PROFILE_ENV_APPLIED_FLAG] delete processEnv[PROFILE_ENV_APPLIED_ID] } @@ -882,6 +996,10 @@ export function applyProviderProfileToProcessEnv( profile: ProviderProfile, options?: { primaryModel?: string }, ): void { + // Capture real shell overrides once (pre-managed only). Do not treat values + // left by a previous profile as shell overrides. + captureShellSelfHostedOverridesIfNeeded() + const { route, compatibilityMode } = resolveProfileCompatibility(profile.provider) const primaryModel = options?.primaryModel ?? getPrimaryModel(profile.model) let profileEnv: ProfileEnv @@ -954,10 +1072,14 @@ export function applyProviderProfileToProcessEnv( route.routeId === 'xiaomi-mimo' || route.routeId === 'xiaomi-mimo-token' ? normalizeXiaomiMimoBaseUrl(profile.baseUrl) ?? profile.baseUrl : profile.baseUrl - const openAIProfileEnv: ProfileEnv = { + let openAIProfileEnv: ProfileEnv = { OPENAI_BASE_URL: normalizedProfileBaseUrl, OPENAI_MODEL: primaryModel, } + openAIProfileEnv = applySelfHostedToolsProfileEnv( + openAIProfileEnv, + profile.selfHostedTools, + ) const isAimlapiProfile = profile.provider === 'aimlapi' || route.routeId === 'aimlapi' || @@ -1081,6 +1203,11 @@ export function applyProviderProfileToProcessEnv( clearProviderProfileEnvFromProcessEnv() Object.assign(process.env, nextEnv) + // Re-apply shell-origin overrides only (not prior profile-managed values). + // clearProviderProfileEnvFromProcessEnv already restores shell flags after + // cleanup; re-apply again after Object.assign so profile-managed values do + // not override a genuine shell export. + restoreShellSelfHostedOverrides() process.env[PROFILE_ENV_APPLIED_FLAG] = '1' process.env[PROFILE_ENV_APPLIED_ID] = profile.id } @@ -1391,11 +1518,14 @@ function buildOpenAICompatibleStartupEnv( if (isCloudflareBaseUrl(activeProfile.baseUrl)) { strictEnv.CLOUDFLARE_API_TOKEN = activeProfile.apiKey } - return applySupportedProfileCustomHeaders(activeProfile, strictEnv) + return applySupportedProfileCustomHeaders( + activeProfile, + applySelfHostedToolsProfileEnv(strictEnv, activeProfile.selfHostedTools), + ) } } - const env: ProfileEnv = { + let env: ProfileEnv = { OPENAI_BASE_URL: activeProfile.baseUrl, OPENAI_MODEL: getPrimaryModel(activeProfile.model), ...(activeProfile.apiFormat ? { OPENAI_API_FORMAT: activeProfile.apiFormat } : {}), @@ -1411,6 +1541,7 @@ function buildOpenAICompatibleStartupEnv( } : {}), } + env = applySelfHostedToolsProfileEnv(env, activeProfile.selfHostedTools) if (isAimlapiProfile) { env.CLAUDE_CODE_PROVIDER_ROUTE_ID = 'aimlapi' @@ -1571,7 +1702,7 @@ function buildStartupProfileFromActiveProfile( processEnv: process.env, }) ?? null return env - ? { profile: 'nvidia-nim', env: applySupportedProfileCustomHeaders(activeProfile, env) } + ? { profile: 'nvidia-nim', env: applySupportedProfileCustomHeaders(activeProfile, applySelfHostedToolsProfileEnv(env, activeProfile.selfHostedTools)) } : null } @@ -1597,7 +1728,7 @@ function buildStartupProfileFromActiveProfile( processEnv: process.env, }) ?? null return env - ? { profile: 'openai', env: applySupportedProfileCustomHeaders(activeProfile, env) } + ? { profile: 'openai', env: applySupportedProfileCustomHeaders(activeProfile, applySelfHostedToolsProfileEnv(env, activeProfile.selfHostedTools)) } : null } @@ -1610,7 +1741,7 @@ function buildStartupProfileFromActiveProfile( processEnv: process.env, }) ?? null return env - ? { profile: 'openai', env: applySupportedProfileCustomHeaders(activeProfile, env) } + ? { profile: 'openai', env: applySupportedProfileCustomHeaders(activeProfile, applySelfHostedToolsProfileEnv(env, activeProfile.selfHostedTools)) } : null } @@ -1623,7 +1754,7 @@ function buildStartupProfileFromActiveProfile( processEnv: process.env, }) ?? null return env - ? { profile: 'openai', env: applySupportedProfileCustomHeaders(activeProfile, env) } + ? { profile: 'openai', env: applySupportedProfileCustomHeaders(activeProfile, applySelfHostedToolsProfileEnv(env, activeProfile.selfHostedTools)) } : null }