From f652c248aad3e587a0176a7788c0559fc5936b1f Mon Sep 17 00:00:00 2001 From: 0xfandom Date: Tue, 19 May 2026 11:59:51 +0530 Subject: [PATCH] fix(recovery): keep thinking blocks on resume for reasoning-echo providers DeepSeek, Moonshot/Kimi, Z.AI GLM, and Xiaomi MiMo require `reasoning_content` echoed back on assistant messages in thinking mode (`preserveReasoningContent` in the openai-shim runtime config). The shim populates that field from the `thinking` content block on the Anthropic-side message, so stripping those blocks during 3P resume left no source and the provider 400'd with: The `reasoning_content` in the thinking mode must be passed back to the API. Skip the 3P thinking strip when the active route/model resolves to a shim config with `preserveReasoningContent: true`. Other 3P providers (generic OpenAI, etc.) keep the original strip from #248 finding 5. Closes #957. --- src/utils/conversationRecovery.test.ts | 61 ++++++++++++++++++++++++++ src/utils/conversationRecovery.ts | 40 ++++++++++++++--- 2 files changed, 94 insertions(+), 7 deletions(-) diff --git a/src/utils/conversationRecovery.test.ts b/src/utils/conversationRecovery.test.ts index fe94001ea1..d688b29f00 100644 --- a/src/utils/conversationRecovery.test.ts +++ b/src/utils/conversationRecovery.test.ts @@ -171,3 +171,64 @@ test('deserializeMessages preserves thinking blocks for GitHub native Claude tra }> expect(content.some(block => block.type === 'thinking')).toBe(true) }) + +test('deserializeMessages preserves thinking blocks for DeepSeek 3P provider (#957)', async () => { + // Regression: DeepSeek requires `reasoning_content` echoed back on assistant + // messages in thinking mode. The shim reads the thinking block to populate + // that field — stripping it on resume left the shim with no source and the + // provider 400'd ("reasoning_content in the thinking mode must be passed + // back"). preserveReasoningContent: true (from runtimeMetadata's DeepSeek + // shim config inference) must opt the provider out of the 3P thinking strip. + clearProviderEnv() + process.env.CLAUDE_CODE_USE_OPENAI = '1' + process.env.OPENAI_BASE_URL = 'https://api.deepseek.com/v1' + process.env.OPENAI_MODEL = 'deepseek-v4-flash' + const { deserializeMessages } = await importFreshConversationRecovery() + + const deserialized = deserializeMessages([ + { + type: 'assistant', + message: { + role: 'assistant', + content: [ + { type: 'thinking', thinking: 'chain of thought' }, + { type: 'text', text: 'answer' }, + ], + }, + } as any, + ]) + + const content = (deserialized[0] as any)?.message?.content as Array<{ + type: string + }> + expect(content.some(block => block.type === 'thinking')).toBe(true) +}) + +test('deserializeMessages still strips thinking blocks for generic OpenAI 3P (no preserveReasoningContent)', async () => { + // Counter-test: providers that don't set preserveReasoningContent keep the + // original strip behaviour from #248 — thinking blocks were causing 400s + // there, and the fix for #957 must not regress that path. + clearProviderEnv() + process.env.CLAUDE_CODE_USE_OPENAI = '1' + process.env.OPENAI_BASE_URL = 'https://api.openai.com/v1' + process.env.OPENAI_MODEL = 'gpt-5-mini' + const { deserializeMessages } = await importFreshConversationRecovery() + + const deserialized = deserializeMessages([ + { + type: 'assistant', + message: { + role: 'assistant', + content: [ + { type: 'thinking', thinking: 'noise' }, + { type: 'text', text: 'answer' }, + ], + }, + } as any, + ]) + + const content = (deserialized[0] as any)?.message?.content as Array<{ + type: string + }> + expect(content.some(block => block.type === 'thinking')).toBe(false) +}) diff --git a/src/utils/conversationRecovery.ts b/src/utils/conversationRecovery.ts index 953216a40d..2af637c1c9 100644 --- a/src/utils/conversationRecovery.ts +++ b/src/utils/conversationRecovery.ts @@ -25,7 +25,10 @@ import { } from './fileHistory.js' import { logError } from './log.js' import { getAPIProvider } from './model/providers.js' -import { usesAnthropicNativeMessageFormat } from '../integrations/runtimeMetadata.js' +import { + resolveOpenAIShimRuntimeContext, + usesAnthropicNativeMessageFormat, +} from '../integrations/runtimeMetadata.js' import { createAssistantMessage, createUserMessage, @@ -198,9 +201,25 @@ function stripThinkingBlocks(messages: NormalizedMessage[]): NormalizedMessage[] }, []) } +// Some 3P providers require `reasoning_content` echoed back on assistant +// messages (DeepSeek thinking mode, Moonshot/Kimi, Z.AI GLM, MiMo, etc.). The +// openai-shim re-shapes those messages and reads the source-of-truth from the +// `thinking` block on the Anthropic side. Stripping the block leaves the shim +// with no reasoning text and the provider 400s with +// "reasoning_content in the thinking mode must be passed back" (issue #957). +// +// Vendors declare this need via `openaiShim.preserveReasoningContent: true` +// in their descriptor, so derive the answer from the resolved shim config +// instead of hardcoding model name prefixes — that automatically covers any +// future vendor that opts in without code changes here. function shouldPreserveThinkingBlocksForProviderReplay(): boolean { - const model = process.env.OPENAI_MODEL?.trim().toLowerCase() ?? '' - return model.startsWith('mimo-v2') + return ( + resolveOpenAIShimRuntimeContext({ + processEnv: process.env, + baseUrl: process.env.OPENAI_BASE_URL, + model: process.env.OPENAI_MODEL, + }).openaiShimConfig.preserveReasoningContent === true + ) } /** @@ -256,6 +275,13 @@ export function deserializeMessagesWithInterruptDetection( // Strip thinking/redacted_thinking content blocks from assistant messages // when resuming against a 3P provider. These Anthropic-specific blocks cause // 400 errors or context corruption on OpenAI-compatible providers (issue #248 finding 5). + // + // Exception: providers that require `reasoning_content` echoed back on + // assistant messages (DeepSeek thinking mode, Moonshot/Kimi, Z.AI GLM, etc.) + // read the source-of-truth from the `thinking` block when re-shaping the + // outgoing OpenAI-format message. Stripping the block leaves the shim with + // no reasoning text to attach, and the provider 400s with + // "reasoning_content in the thinking mode must be passed back" (issue #957). const provider = getAPIProvider() const isAnthropicNativeTransport = usesAnthropicNativeMessageFormat({ processEnv: process.env, @@ -264,10 +290,10 @@ export function deserializeMessagesWithInterruptDetection( }) const isThirdPartyProvider = provider !== 'foundry' && !isAnthropicNativeTransport - const thinkingStripped = isThirdPartyProvider - && !shouldPreserveThinkingBlocksForProviderReplay() - ? stripThinkingBlocks(filteredThinking) - : filteredThinking + const thinkingStripped = + isThirdPartyProvider && !shouldPreserveThinkingBlocksForProviderReplay() + ? stripThinkingBlocks(filteredThinking) + : filteredThinking // Filter out assistant messages with only whitespace text content. // This can happen when model outputs "\n\n" before thinking, user cancels mid-stream.