Skip to content
Closed
61 changes: 59 additions & 2 deletions src/integrations/runtimeMetadata.ts
Original file line number Diff line number Diff line change
Expand Up @@ -92,10 +92,19 @@ function mergeOpenAIShimConfig(
entryConfig: Partial<OpenAIShimTransportConfig> | undefined,
inferredConfig: Partial<OpenAIShimTransportConfig> | undefined,
): OpenAIShimTransportConfig {
const descriptorMaxTokensField =
entryConfig?.maxTokensField ?? baseConfig?.maxTokensField

return {
...baseConfig,
...entryConfig,
...inferredConfig,
// Preserve descriptor-level maxTokensField over model-inferred value.
// Direct providers (e.g. Xiaomi) declare their own token field in the
// descriptor; model-name inference should not override it.
...(descriptorMaxTokensField !== undefined
? { maxTokensField: descriptorMaxTokensField }
: {}),
removeBodyFields: mergeRemoveBodyFields(
baseConfig?.removeBodyFields,
entryConfig?.removeBodyFields,
Expand Down Expand Up @@ -163,6 +172,28 @@ function inferRemoteModelOpenAIShimConfig(
}
}

if (normalizedModel.includes('mimo')) {
return {
preserveReasoningContent: true,
requireReasoningContentOnAssistantMessages: true,
reasoningContentFallback: '',
thinkingRequestFormat: 'deepseek-compatible',
maxTokensField: 'max_completion_tokens',
removeBodyFields: ['store'],
}
}

if (normalizedModel.includes('glm')) {
return {
preserveReasoningContent: true,
requireReasoningContentOnAssistantMessages: true,
reasoningContentFallback: '',
thinkingRequestFormat: 'deepseek-compatible',
maxTokensField: 'max_tokens',
removeBodyFields: ['store'],
}
}

return undefined
}

Expand Down Expand Up @@ -215,12 +246,38 @@ export function resolveOpenAIShimRuntimeContext(options?: {
descriptor && routeId
? getCatalogEntryForModel(routeId, options?.model)
: null
const remoteModelInferredConfig = inferRemoteModelOpenAIShimConfig(options?.model)
const inferredConfig =
options?.treatAsLocal === true
? {
maxTokensField: 'max_tokens' as const,
// Don't hardcode maxTokensField here — let the descriptor (via
// mergeOpenAIShimConfig) or the model-inferred value take precedence.
// Hardcoding 'max_tokens' breaks providers like MiMo that require
// 'max_completion_tokens', even when accessed through a local proxy.
...(remoteModelInferredConfig?.maxTokensField !== undefined
? { maxTokensField: remoteModelInferredConfig.maxTokensField }
: { maxTokensField: 'max_tokens' as const }),
// Local proxies (e.g. key routers like grouter) may forward to remote
// reasoning models that require reasoning_content. Preserve the
// model-inferred config for reasoning fields so local routing doesn't
// strip thinking continuity.
...(remoteModelInferredConfig?.preserveReasoningContent !== undefined
? { preserveReasoningContent: remoteModelInferredConfig.preserveReasoningContent }
: {}),
...(remoteModelInferredConfig?.requireReasoningContentOnAssistantMessages !== undefined
? { requireReasoningContentOnAssistantMessages: remoteModelInferredConfig.requireReasoningContentOnAssistantMessages }
: {}),
...(remoteModelInferredConfig?.reasoningContentFallback !== undefined
? { reasoningContentFallback: remoteModelInferredConfig.reasoningContentFallback }
: {}),
...(remoteModelInferredConfig?.thinkingRequestFormat !== undefined
? { thinkingRequestFormat: remoteModelInferredConfig.thinkingRequestFormat }
: {}),
...(remoteModelInferredConfig?.removeBodyFields !== undefined
? { removeBodyFields: remoteModelInferredConfig.removeBodyFields }
: {}),
}
: inferRemoteModelOpenAIShimConfig(options?.model)
: remoteModelInferredConfig

return {
routeId,
Expand Down
49 changes: 43 additions & 6 deletions src/services/api/openaiShim.ts
Original file line number Diff line number Diff line change
Expand Up @@ -487,11 +487,13 @@ function convertMessages(
system: unknown,
options?: {
preserveReasoningContent?: boolean
requireReasoningContentOnAssistantMessages?: boolean
reasoningContentFallback?: '' | 'omit'
preserveGeminiThoughtSignature?: boolean
},
): OpenAIMessage[] {
const preserveReasoningContent = options?.preserveReasoningContent === true
const requireReasoningContent = options?.requireReasoningContentOnAssistantMessages === true
const reasoningContentFallback = options?.reasoningContentFallback
const preserveGeminiThoughtSignature = options?.preserveGeminiThoughtSignature === true
const result: OpenAIMessage[] = []
Expand Down Expand Up @@ -608,12 +610,14 @@ function convertMessages(
const thinkingText = (thinkingBlock as { thinking?: string } | undefined)?.thinking
if (typeof thinkingText === 'string' && thinkingText.trim().length > 0) {
assistantMsg.reasoning_content = thinkingText
} else if (
toolUses.length > 0 &&
reasoningContentFallback === ''
) {
} else if (reasoningContentFallback === '') {
assistantMsg.reasoning_content = ''
}
} else if (requireReasoningContent) {
// Provider requires reasoning_content on every assistant message but
// preserveReasoningContent is not set (no thinking block to echo).
// Attach empty string to satisfy API validation.
assistantMsg.reasoning_content = ''
}

if (toolUses.length > 0) {
Expand Down Expand Up @@ -699,6 +703,14 @@ function convertMessages(
})(),
}

// For providers that require reasoning_content on every assistant message,
// attach an empty string when no thinking block is present.
if (preserveReasoningContent && reasoningContentFallback === '') {
assistantMsg.reasoning_content = ''
} else if (requireReasoningContent) {
assistantMsg.reasoning_content = ''
}

if (assistantMsg.content) {
result.push(assistantMsg)
}
Expand All @@ -720,10 +732,18 @@ function convertMessages(
// assistant response to satisfy the strict role sequence:
// ... -> assistant (calls) -> tool (results) -> assistant (semantic) -> user (next)
if (prev && prev.role === 'tool' && msg.role === 'user') {
coalesced.push({
const syntheticMsg: OpenAIMessage = {
role: 'assistant',
content: '[Tool execution interrupted by user]',
})
}
// Providers that require reasoning_content on every assistant message
// need it on this synthetic message too, or the API returns 400.
if (preserveReasoningContent && reasoningContentFallback === '') {
syntheticMsg.reasoning_content = ''
} else if (requireReasoningContent) {
syntheticMsg.reasoning_content = ''
}
coalesced.push(syntheticMsg)
}

const lastAfterPossibleInjection = coalesced[coalesced.length - 1]
Expand Down Expand Up @@ -766,6 +786,22 @@ function convertMessages(
...msg.tool_calls,
]
}

// Preserve reasoning_content when coalescing assistant messages.
// Never overwrite a non-empty reasoning_content with an empty one,
// since empty values are fallback placeholders while non-empty ones
// carry actual thinking text that providers require.
if (
msg.role === 'assistant' &&
msg.reasoning_content !== undefined &&
lastAfterPossibleInjection.role === 'assistant'
) {
const existing = lastAfterPossibleInjection.reasoning_content
const incoming = msg.reasoning_content
if (typeof existing !== 'string' || existing.length === 0 || incoming.length > 0) {
lastAfterPossibleInjection.reasoning_content = incoming
}
}
} else {
coalesced.push(msg)
}
Expand Down Expand Up @@ -1782,6 +1818,7 @@ class OpenAIShimMessages {
const shimConfig = runtimeShimContext.openaiShimConfig
const openaiMessages = convertMessages(compressedMessages, params.system, {
preserveReasoningContent: shimConfig.preserveReasoningContent,
requireReasoningContentOnAssistantMessages: shimConfig.requireReasoningContentOnAssistantMessages,
reasoningContentFallback: shimConfig.reasoningContentFallback,
preserveGeminiThoughtSignature: shouldPreserveGeminiThoughtSignature(
request.resolvedModel,
Expand Down