From a4c463bac2ef50179897a3707f40a94aa5b6c97c Mon Sep 17 00:00:00 2001 From: Xiangzhe Date: Sat, 18 Jul 2026 14:43:35 +0800 Subject: [PATCH] fix(kimi-coding): capture and replay reasoning for thinking-mode turns Kimi Coding (claude-format upstream) never engaged reasoning replay: requiresReasoningReplay() had no kimi-coding/kimi-coding-apikey provider entry and only matched /kimi-k2/i model ids, so thinking was neither captured nor re-injected on multi-turn requests. Additionally, streamed Claude thinking_delta chunks were accumulated into content instead of accumulatedReasoning in createSSEStream, so the reconstructed completion body carried no reasoning_content for the cache to capture. - reasoningCache: add kimi-coding/kimi-coding-apikey providers; broaden model pattern to /kimi[-/]k\d/i (covers k2.6/k2.7 incl. namespaced ids, excludes kimi-latest and non-thinking aliases) - stream: accumulate Claude delta.thinking into accumulatedReasoning so the completion body exposes reasoning_content for replay capture - tests: provider/model predicate cases + a reconstructed-stream-body regression test separating thinking from visible text - docs: sync REASONING_REPLAY provider/pattern lists --- docs/routing/REASONING_REPLAY.md | 4 +- open-sse/services/reasoningCache.ts | 9 +- open-sse/utils/stream.ts | 8 +- .../claude-stream-reconstructed-body.test.ts | 86 +++++++++++++ tests/unit/service-reasoning-cache.test.ts | 117 ++++++++++++++++-- 5 files changed, 205 insertions(+), 19 deletions(-) create mode 100644 tests/unit/claude-stream-reconstructed-body.test.ts diff --git a/docs/routing/REASONING_REPLAY.md b/docs/routing/REASONING_REPLAY.md index bf56c0d8b3fb..c83f1edc5f32 100644 --- a/docs/routing/REASONING_REPLAY.md +++ b/docs/routing/REASONING_REPLAY.md @@ -89,6 +89,8 @@ Replay is enabled when `requiresReasoningReplay(provider, model)` returns `true` - `sambanova` - `fireworks` - `together` +- `kimi-coding` +- `kimi-coding-apikey` - `xiaomi-mimo` **Model regex patterns (case-insensitive):** @@ -98,7 +100,7 @@ Replay is enabled when `requiresReasoningReplay(provider, model)` returns `true` - `/deepseek-chat/i` - `/deepseek[-/]?v4[-.]flash/i` and `/deepseek[-/]?v4[-.]pro/i` (V4 Flash / Pro, optional `-free` suffix) - `/(deepseek|zen\/deepseek)-v4/i` -- `/kimi-k2/i` +- `/kimi[-/]k\d/i` - `/qwq/i` - `/qwen.*think/i` - `/glm.*think/i` diff --git a/open-sse/services/reasoningCache.ts b/open-sse/services/reasoningCache.ts index 9c7cabd29ca6..3d4a715fe019 100644 --- a/open-sse/services/reasoningCache.ts +++ b/open-sse/services/reasoningCache.ts @@ -34,6 +34,10 @@ const REASONING_REPLAY_PROVIDERS = new Set([ "sambanova", "fireworks", "together", + // Kimi Coding thinking-mode upstreams require reasoning_content replay under + // the same strict multi-turn contract as DeepSeek. + "kimi-coding", + "kimi-coding-apikey", // Xiaomi MiMo enforces the same "pass back reasoning_content on subsequent // turns" contract as DeepSeek/Kimi-thinking. Without replay the upstream // 400s with "Param Incorrect: The reasoning_content in the thinking mode @@ -47,8 +51,9 @@ const REASONING_REPLAY_MODEL_PATTERNS = [ /deepseek-chat/i, /deepseek[-/]v4[-.](flash|pro)(-free)?/i, /zen\/deepseek-v4/i, - /kimi-k2/i, - /kimi-k3/i, + // Match native kimi-kN and namespaced kimi/kN families without treating + // generic aliases such as kimi-latest as strict thinking models. + /kimi[-/]k\d/i, /qwq/i, /qwen.*think/i, /glm.*think/i, diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index e8356db21bb4..cc563ef3f2eb 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -1567,8 +1567,8 @@ export function createSSEStream(options: StreamOptions = {}) { } if (parsed.delta?.thinking) { totalContentLength += parsed.delta.thinking.length; - passthroughAccumulatedContent = appendBoundedText( - passthroughAccumulatedContent, + passthroughAccumulatedReasoning = appendBoundedText( + passthroughAccumulatedReasoning, parsed.delta.thinking ); } @@ -1944,8 +1944,8 @@ export function createSSEStream(options: StreamOptions = {}) { if (parsed.delta?.thinking) { const t = parsed.delta.thinking; totalContentLength += t.length; - if (state?.accumulatedContent !== undefined && typeof t === "string") - state.accumulatedContent = appendBoundedText(state.accumulatedContent, t); + if (state?.accumulatedReasoning !== undefined && typeof t === "string") + state.accumulatedReasoning = appendBoundedText(state.accumulatedReasoning, t); } // OpenAI format diff --git a/tests/unit/claude-stream-reconstructed-body.test.ts b/tests/unit/claude-stream-reconstructed-body.test.ts new file mode 100644 index 000000000000..af7a7982dffb --- /dev/null +++ b/tests/unit/claude-stream-reconstructed-body.test.ts @@ -0,0 +1,86 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { createSSEStream } from "../../open-sse/utils/stream.ts"; +import { FORMATS } from "../../open-sse/translator/formats.ts"; + +async function processClaudeStream(events: Record[]) { + let responseBody: unknown; + const transform = createSSEStream({ + sourceFormat: FORMATS.OPENAI, + targetFormat: FORMATS.CLAUDE, + model: "kimi-for-coding", + onComplete: (result) => { + responseBody = result.responseBody; + }, + }) as TransformStream; + + const writer = transform.writable.getWriter(); + const reader = transform.readable.getReader(); + const encoder = new TextEncoder(); + const drain = (async () => { + while (!(await reader.read()).done) {} + })(); + + for (const event of events) { + await writer.write(encoder.encode(`data: ${JSON.stringify(event)}\n\n`)); + } + await writer.close(); + await drain; + + return responseBody as { + choices: Array<{ + message: { + content: string | null; + reasoning_content?: string; + tool_calls?: unknown[]; + }; + }>; + }; +} + +test("reconstructed completion separates Claude thinking from visible text", async () => { + const responseBody = await processClaudeStream([ + { + type: "message_start", + message: { id: "msg_test", type: "message", role: "assistant", content: [] }, + }, + { + type: "content_block_start", + index: 0, + content_block: { type: "thinking", thinking: "" }, + }, + { type: "content_block_delta", index: 0, delta: { type: "thinking_delta", thinking: "plan " } }, + { + type: "content_block_delta", + index: 0, + delta: { type: "thinking_delta", thinking: "carefully" }, + }, + { type: "content_block_stop", index: 0 }, + { type: "content_block_start", index: 1, content_block: { type: "text", text: "" } }, + { + type: "content_block_delta", + index: 1, + delta: { type: "text_delta", text: "Visible answer" }, + }, + { type: "content_block_stop", index: 1 }, + { + type: "content_block_start", + index: 2, + content_block: { type: "tool_use", id: "tool_1", name: "lookup", input: {} }, + }, + { + type: "content_block_delta", + index: 2, + delta: { type: "input_json_delta", partial_json: '{"query":"test"}' }, + }, + { type: "content_block_stop", index: 2 }, + { type: "message_delta", delta: { stop_reason: "tool_use" }, usage: { output_tokens: 12 } }, + { type: "message_stop" }, + ]); + + const message = responseBody.choices[0].message; + assert.equal(message.reasoning_content, "plan carefully"); + assert.equal(message.content, "Visible answer"); + assert.equal(message.tool_calls?.length, 1); +}); diff --git a/tests/unit/service-reasoning-cache.test.ts b/tests/unit/service-reasoning-cache.test.ts index 3474ddb4db30..7b3a2490dfaf 100644 --- a/tests/unit/service-reasoning-cache.test.ts +++ b/tests/unit/service-reasoning-cache.test.ts @@ -6,47 +6,140 @@ const mod = await import("../../open-sse/services/reasoningCache.ts"); describe("reasoningCache helpers", () => { describe("isDeepSeekReasoningModel", () => { it("returns true for deepseek-v4 models with thinking enabled", () => { - assert.equal(mod.isDeepSeekReasoningModel({ provider: "deepseek", model: "deepseek-v4-flash", thinkingEnabled: true }), true); - assert.equal(mod.isDeepSeekReasoningModel({ provider: "deepseek", model: "deepseek/v4-pro", thinkingEnabled: true }), true); + assert.equal( + mod.isDeepSeekReasoningModel({ + provider: "deepseek", + model: "deepseek-v4-flash", + thinkingEnabled: true, + }), + true + ); + assert.equal( + mod.isDeepSeekReasoningModel({ + provider: "deepseek", + model: "deepseek/v4-pro", + thinkingEnabled: true, + }), + true + ); }); it("returns false without thinkingEnabled", () => { - assert.equal(mod.isDeepSeekReasoningModel({ provider: "deepseek", model: "deepseek-v4-flash" }), false); - assert.equal(mod.isDeepSeekReasoningModel({ provider: "deepseek", model: "deepseek-v4-flash", thinkingEnabled: false }), false); + assert.equal( + mod.isDeepSeekReasoningModel({ provider: "deepseek", model: "deepseek-v4-flash" }), + false + ); + assert.equal( + mod.isDeepSeekReasoningModel({ + provider: "deepseek", + model: "deepseek-v4-flash", + thinkingEnabled: false, + }), + false + ); }); it("returns false for non-v4 models", () => { - assert.equal(mod.isDeepSeekReasoningModel({ provider: "deepseek", model: "deepseek-chat", thinkingEnabled: true }), false); + assert.equal( + mod.isDeepSeekReasoningModel({ + provider: "deepseek", + model: "deepseek-chat", + thinkingEnabled: true, + }), + false + ); }); }); describe("requiresReasoningReplay", () => { it("returns true for reasoning_content interleaved field", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "any", model: "any", interleavedField: "reasoning_content" }), true); + assert.equal( + mod.requiresReasoningReplay({ + provider: "any", + model: "any", + interleavedField: "reasoning_content", + }), + true + ); }); it("returns false for reasoning_details interleaved field", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "any", model: "any", interleavedField: "reasoning_details" }), false); + assert.equal( + mod.requiresReasoningReplay({ + provider: "any", + model: "any", + interleavedField: "reasoning_details", + }), + false + ); }); it("returns false for deepseek-reasoner", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "deepseek", model: "deepseek-reasoner" }), false); + assert.equal( + mod.requiresReasoningReplay({ provider: "deepseek", model: "deepseek-reasoner" }), + false + ); }); it("returns false for deepseek-r1", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "deepseek", model: "deepseek-r1" }), false); + assert.equal( + mod.requiresReasoningReplay({ provider: "deepseek", model: "deepseek-r1" }), + false + ); }); it("returns true for DeepSeek V4 thinking models", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "deepseek", model: "deepseek-v4-flash", thinkingEnabled: true }), true); + assert.equal( + mod.requiresReasoningReplay({ + provider: "deepseek", + model: "deepseek-v4-flash", + thinkingEnabled: true, + }), + true + ); }); it("returns true for known replay providers", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "deepseek", model: "some-model" }), true); + assert.equal( + mod.requiresReasoningReplay({ provider: "deepseek", model: "some-model" }), + true + ); + }); + + it("returns true for Kimi Coding providers regardless of model alias", () => { + assert.equal(mod.requiresReasoningReplay({ provider: "kimi-coding", model: "k3" }), true); + assert.equal( + mod.requiresReasoningReplay({ provider: "kimi-coding-apikey", model: "kimi-k2.6" }), + true + ); + }); + + it("detects native Kimi thinking model IDs without matching unrelated aliases", () => { + for (const model of [ + "kimi-k2", + "kimi-k2.6", + "kimi-k2.6-thinking", + "kimi-k2.7-code", + "kimi-k2.7-code-highspeed", + "moonshotai/kimi-k2.7-code", + ]) { + assert.equal(mod.requiresReasoningReplay({ provider: "some-other", model }), true, model); + } + + for (const model of ["k3", "moonshot-v1-8k", "kimi-latest"]) { + assert.equal(mod.requiresReasoningReplay({ provider: "some-other", model }), false, model); + } }); it("returns false when allowLegacyFallback is false and no explicit signal", () => { - assert.equal(mod.requiresReasoningReplay({ provider: "unknown", model: "unknown", allowLegacyFallback: false }), false); + assert.equal( + mod.requiresReasoningReplay({ + provider: "unknown", + model: "unknown", + allowLegacyFallback: false, + }), + false + ); }); });