Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions docs/advanced-setup.md
Original file line number Diff line number Diff line change
Expand Up @@ -225,6 +225,26 @@ export OPENAI_BASE_URL=http://localhost:1234/v1
export OPENAI_MODEL=your-model-name
```

### llama.cpp / vLLM / other local OpenAI-compatible servers

```bash
export CLAUDE_CODE_USE_OPENAI=1
export OPENAI_BASE_URL=http://127.0.0.1:8080/v1
export OPENAI_MODEL=/models/your-model.gguf
```

OpenClaude never inserts the Mistral/Devstral tool-boundary placeholder
(`[Tool results received]`) for local OpenAI-compatible hosts (loopback,
RFC1918, CGNAT/`100.64/10`, `host.docker.internal`, reserved TLDs such as
`.local` / `.localhost` / `.home.arpa` / `.lan`) or Ollama endpoints. That
filler is reserved for Mistral cloud routes, `CLAUDE_CODE_USE_MISTRAL=1` on
non-local endpoints, and non-local model ids that clearly look Mistral-class,
so quantized local models do not imitate it and end the turn early.

If a local model still echoes short repeated phrases, enable server-side
`repeat_penalty` / DRY sampling (llama.cpp defaults often leave
`--repeat-penalty` at `1.0` and `dry_multiplier` at `0.0`).

### Together AI

```bash
Expand Down
13 changes: 13 additions & 0 deletions src/__tests__/bugfixes.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -111,6 +111,19 @@ describe('Agent loop continuation nudge', () => {
// Transition intent detected (requires explicit action verb or transition phrase)
expect(analyzeContinuationIntent("So now I will start task 2").shouldNudge).toBe(true)
expect(analyzeContinuationIntent("I will now do the following").shouldNudge).toBe(true)

// Model echoed the OpenAI-shim transport placeholder as its entire reply
expect(analyzeContinuationIntent('[Tool results received]')).toEqual({
shouldNudge: true,
reason: 'tool_result_placeholder_echo',
})
expect(analyzeContinuationIntent(' [Tool results received] ').shouldNudge).toBe(true)
expect(analyzeContinuationIntent('[Tool results received].').shouldNudge).toBe(true)
expect(analyzeContinuationIntent('"[Tool results received]"').shouldNudge).toBe(true)
expect(analyzeContinuationIntent('[Tool results received],').shouldNudge).toBe(true)
expect(
analyzeContinuationIntent('Done. [Tool results received] and more text').shouldNudge,
).toBe(false)

// Completion marker suppresses nudge
expect(analyzeContinuationIntent("Task finished").shouldNudge).toBe(false)
Expand Down
99 changes: 96 additions & 3 deletions src/services/api/openaiShim.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,7 @@ const originalEnv = {
GH_TOKEN: process.env.GH_TOKEN,
CLAUDE_CODE_USE_OPENAI: process.env.CLAUDE_CODE_USE_OPENAI,
CLAUDE_CODE_USE_GEMINI: process.env.CLAUDE_CODE_USE_GEMINI,
CLAUDE_CODE_USE_MISTRAL: process.env.CLAUDE_CODE_USE_MISTRAL,
GEMINI_API_KEY: process.env.GEMINI_API_KEY,
GOOGLE_API_KEY: process.env.GOOGLE_API_KEY,
GEMINI_ACCESS_TOKEN: process.env.GEMINI_ACCESS_TOKEN,
Expand Down Expand Up @@ -553,6 +554,7 @@ afterEach(() => {
restoreEnv('GH_TOKEN', originalEnv.GH_TOKEN)
restoreEnv('CLAUDE_CODE_USE_OPENAI', originalEnv.CLAUDE_CODE_USE_OPENAI)
restoreEnv('CLAUDE_CODE_USE_GEMINI', originalEnv.CLAUDE_CODE_USE_GEMINI)
restoreEnv('CLAUDE_CODE_USE_MISTRAL', originalEnv.CLAUDE_CODE_USE_MISTRAL)
restoreEnv('GEMINI_API_KEY', originalEnv.GEMINI_API_KEY)
restoreEnv('GOOGLE_API_KEY', originalEnv.GOOGLE_API_KEY)
restoreEnv('GEMINI_ACCESS_TOKEN', originalEnv.GEMINI_ACCESS_TOKEN)
Expand Down Expand Up @@ -8359,11 +8361,10 @@ test('preserves valid tool_result and drops orphan tool_result', async () => {
const orphanMessage = toolMessages.find(m => m.tool_call_id === 'orphan_call_2')
expect(orphanMessage).toBeUndefined()

// Actually, the semantic message IS injected here because the user block with orphan
// tool result is converted to:
// Mistral model name triggers the Mistral-only tool→user semantic boundary.
// 1. Tool result (valid_call_1) -> role 'tool'
// 2. User content ("What happened?") -> role 'user'
// This triggers the tool -> assistant injection.
// This triggers the tool -> assistant injection for Mistral.
const assistantMessages = messages.filter(m => m.role === 'assistant')
expect(assistantMessages.some(m => m.content === '[Tool results received]')).toBe(true)
})
Expand Down Expand Up @@ -8498,6 +8499,98 @@ test('injects semantic assistant message when tool result is followed by user me
expect(semanticMsg.content).not.toContain('interrupted')
expect(semanticMsg.content).not.toContain('user')
})

test('does not inject semantic assistant message for non-Mistral OpenAI-compatible models', async () => {
let requestBody: Record<string, unknown> | undefined

globalThis.fetch = (async (_input, init) => {
requestBody = JSON.parse(String(init?.body))
return new Response(JSON.stringify({
id: 'chatcmpl-3',
object: 'chat.completion',
created: 123456789,
model: 'Qwen3.6-35B-A3B',
choices: [{ message: { role: 'assistant', content: 'hi' }, finish_reason: 'stop' }],
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }
}), { headers: { 'Content-Type': 'application/json' } })
}) as unknown as FetchType

process.env.OPENAI_BASE_URL = 'http://127.0.0.1:8080/v1'
process.env.OPENAI_API_KEY = 'local'
process.env.OPENAI_MODEL = 'Qwen3.6-35B-A3B'
delete process.env.CLAUDE_CODE_USE_MISTRAL

const client = createOpenAIShimClient({}) as OpenAIShimClient

await client.beta.messages.create({
model: 'Qwen3.6-35B-A3B',
messages: [
{
role: 'assistant',
content: [{ type: 'tool_use', id: 'call_1', name: 'search', input: {} }],
},
{
role: 'user',
content: [{ type: 'tool_result', tool_use_id: 'call_1', content: 'Result' }],
},
{ role: 'user', content: 'Next user query' },
],
max_tokens: 64,
stream: false,
})

const messages = requestBody?.messages as Array<Record<string, unknown>>
expect(messages.map(m => m.role)).toEqual(['assistant', 'tool', 'user'])
expect(messages.some(m => m.content === '[Tool results received]')).toBe(false)
})

test('providerOverride does not inherit parent CLAUDE_CODE_USE_MISTRAL for semantic boundary', async () => {
let requestBody: Record<string, unknown> | undefined

globalThis.fetch = (async (_input, init) => {
requestBody = JSON.parse(String(init?.body))
return new Response(JSON.stringify({
id: 'chatcmpl-4',
object: 'chat.completion',
created: 123456789,
model: 'Qwen3.6-35B-A3B',
choices: [{ message: { role: 'assistant', content: 'hi' }, finish_reason: 'stop' }],
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }
}), { headers: { 'Content-Type': 'application/json' } })
}) as unknown as FetchType

process.env.CLAUDE_CODE_USE_MISTRAL = '1'
process.env.OPENAI_API_KEY = 'remote-key'

const client = createOpenAIShimClient({
providerOverride: {
model: 'Qwen3.6-35B-A3B',
baseURL: 'https://api.example.com/v1',
apiKey: 'override-remote',
},
}) as OpenAIShimClient

await client.beta.messages.create({
model: 'mistral-large-latest',
messages: [
{
role: 'assistant',
content: [{ type: 'tool_use', id: 'call_1', name: 'search', input: {} }],
},
{
role: 'user',
content: [{ type: 'tool_result', tool_use_id: 'call_1', content: 'Result' }],
},
{ role: 'user', content: 'Next user query' },
],
max_tokens: 64,
stream: false,
})

const messages = requestBody?.messages as Array<Record<string, unknown>>
expect(messages.map(m => m.role)).toEqual(['assistant', 'tool', 'user'])
expect(messages.some(m => m.content === '[Tool results received]')).toBe(false)
})
// openaiShim test extraction seam 136 end


Expand Down
20 changes: 19 additions & 1 deletion src/services/api/openaiShim.ts
Original file line number Diff line number Diff line change
Expand Up @@ -107,6 +107,8 @@ import {
resolveRuntimeCodexCredentials,
resolveProviderRequest,
shouldAttemptLocalToollessRetry,
shouldInjectToolResultSemanticBoundary,
TOOL_RESULT_SEMANTIC_PLACEHOLDER,
type LocalFastPathConfig,
} from './providerConfig.js'
import {
Expand Down Expand Up @@ -765,6 +767,8 @@ function convertMessages(
reasoningContentFallback?: '' | 'omit'
preserveGeminiThoughtSignature?: boolean
supportsImageInputs?: boolean
injectToolResultSemanticBoundary?: boolean
toolResultSemanticPlaceholder?: string
},
): OpenAIMessage[] {
return convertAnthropicMessages(messages, system, {
Expand Down Expand Up @@ -1042,8 +1046,16 @@ class OpenAIShimMessages {
params: ShimCreateParams,
options?: { signal?: AbortSignal; headers?: Record<string, string> },
) {
// A provider override is a complete route, so it must not inherit an
// Azure-style escape hatch or Mistral selector intended for the parent.
// Otherwise a Mistral parent still injects [Tool results received] into a
// Qwen/llama override and reintroduces the post-tool stall (#2039/#2059).
const requestProcessEnv = this.providerOverride
? { ...process.env, OPENAI_AZURE_STYLE: undefined }
? {
...process.env,
OPENAI_AZURE_STYLE: undefined,
CLAUDE_CODE_USE_MISTRAL: undefined,
}
: process.env
return createShimRequest(params, options, {
providerOverride: this.providerOverride,
Expand Down Expand Up @@ -1281,6 +1293,12 @@ class OpenAIShimMessages {
request.baseUrl,
),
supportsImageInputs: shimConfig.supportsImageInputs,
injectToolResultSemanticBoundary: shouldInjectToolResultSemanticBoundary({
baseUrl: request.baseUrl,
model: request.resolvedModel,
processEnv: requestProcessEnv,
}),
toolResultSemanticPlaceholder: TOOL_RESULT_SEMANTIC_PLACEHOLDER,
}),
)

Expand Down
76 changes: 74 additions & 2 deletions src/services/api/openaiShim/messageConversion.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -136,7 +136,8 @@ test('preserves valid tool_result and drops orphan tool_result', () => {

expect(tools).toHaveLength(1)
expect(tools[0]?.tool_call_id).toBe('valid_call_1')
expect(messages.some(message => message.content === '[Tool results received]')).toBe(true)
// Default conversion must not inject the Mistral-only semantic placeholder.
expect(messages.some(message => message.content === '[Tool results received]')).toBe(false)
expect(logs).toContain('Dropping orphan tool_result for ID: orphan_call_2 to prevent API error')
})

Expand All @@ -160,7 +161,7 @@ test('drops empty assistant message when only redacted_thinking block was presen
expect(messages).toEqual([{ role: 'user', content: 'Initial\nInterrupting query' }])
})

test('injects semantic assistant message when tool result is followed by user message', () => {
test('does not inject semantic assistant message between tool and user by default', () => {
const messages = convert([
{
role: 'assistant',
Expand All @@ -173,11 +174,82 @@ test('injects semantic assistant message when tool result is followed by user me
{ role: 'user', content: 'Next user query' },
])

expect(messages.map(message => message.role)).toEqual(['assistant', 'tool', 'user'])
expect(messages.some(message => message.content === '[Tool results received]')).toBe(false)
})

test('injects semantic assistant message when tool result is followed by user message and opted in', () => {
const messages = convertMessages(
[
{
role: 'assistant',
content: [{ type: 'tool_use', id: 'call_1', name: 'search', input: {} }],
},
{
role: 'user',
content: [{ type: 'tool_result', tool_use_id: 'call_1', content: 'Result' }],
},
{ role: 'user', content: 'Next user query' },
],
'',
{ injectToolResultSemanticBoundary: true },
)

expect(messages.map(message => message.role)).toEqual(['assistant', 'tool', 'assistant', 'user'])
expect(messages[2]?.content).toBe('[Tool results received]')
expect(messages[2]?.content).not.toContain('interrupted')
})

test('default conversion keeps tool then snip reminder without placeholder', () => {
const messages = convert([
{
role: 'assistant',
content: [{ type: 'tool_use', id: 'call_1', name: 'Bash', input: { command: 'pwd' } }],
},
{
role: 'user',
content: [{ type: 'tool_result', tool_use_id: 'call_1', content: '/tmp' }],
},
{
role: 'user',
content: '<system-reminder>snip_id=abc123</system-reminder>',
},
])

expect(messages.map(message => message.role)).toEqual(['assistant', 'tool', 'user'])
expect(messages.some(message => message.content === '[Tool results received]')).toBe(false)
expect(String(messages[2]?.content)).toContain('snip_id=abc123')
})

test('strips prior placeholder-only assistant echoes from converted history', () => {
const messages = convert([
{ role: 'user', content: 'Start' },
{ role: 'assistant', content: '[Tool results received]' },
{ role: 'user', content: 'Continue the task' },
])

expect(messages.map(message => message.role)).toEqual(['user'])
expect(String(messages[0]?.content)).toContain('Start')
expect(String(messages[0]?.content)).toContain('Continue the task')
expect(messages.some(message => message.content === '[Tool results received]')).toBe(false)
})

test('strips punctuated and quoted placeholder-only assistant echoes from converted history', () => {
for (const echo of [
'[Tool results received].',
'"[Tool results received]"',
'[tool results received]',
]) {
const messages = convert([
{ role: 'user', content: 'Start' },
{ role: 'assistant', content: echo },
{ role: 'user', content: 'Continue the task' },
])
expect(messages.map(message => message.role)).toEqual(['user'])
expect(messages.some(message => String(message.content ?? '').includes('Tool results'))).toBe(false)
}
})

test('collapses multiple text blocks in tool_result to string for DeepSeek compatibility (issue #774)', () => {
const messages = convert(toolExchange([
{ type: 'text', text: 'line one' },
Expand Down
Loading