From 3eee08499e5dcf909a2211ecb8ea8c4a27dc059f Mon Sep 17 00:00:00 2001 From: Luca Steeb Date: Wed, 14 Jan 2026 18:37:29 +0000 Subject: [PATCH] fix(gateway): cache thought_signature with correct ID MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The gateway was generating tool_call IDs in two places using Date.now(), causing a mismatch between the ID sent to clients and the ID used for Redis caching. This broke multi-turn conversations with Gemini 3 models that require thought_signature. - Add Redis caching in transform-streaming-to-openai.ts where the correct ID is generated - Remove duplicate caching from extract-tool-calls.ts 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude --- .../src/chat/tools/extract-tool-calls.ts | 21 ++---------------- .../tools/transform-streaming-to-openai.ts | 22 ++++++++++++++++++- 2 files changed, 23 insertions(+), 20 deletions(-) diff --git a/apps/gateway/src/chat/tools/extract-tool-calls.ts b/apps/gateway/src/chat/tools/extract-tool-calls.ts index dadf70e106..b941a0c138 100644 --- a/apps/gateway/src/chat/tools/extract-tool-calls.ts +++ b/apps/gateway/src/chat/tools/extract-tool-calls.ts @@ -1,6 +1,3 @@ -import { redisClient } from "@llmgateway/cache"; -import { logger } from "@llmgateway/logger"; - import type { Provider } from "@llmgateway/models"; /** @@ -45,6 +42,8 @@ export function extractToolCalls(data: any, provider: Provider): any[] | null { case "google-vertex": { // Google AI Studio tool calls in streaming // Include thoughtSignature if present (required for Gemini 3 multi-turn conversations) + // Note: Redis caching of thought_signature happens in transform-streaming-to-openai.ts + // where the actual tool_call ID sent to clients is generated const parts = data.candidates?.[0]?.content?.parts || []; return ( parts @@ -65,22 +64,6 @@ export function extractToolCalls(data: any, provider: Provider): any[] | null { thought_signature: part.thoughtSignature, }, }; - // Cache thoughtSignature in Redis for server-side retrieval in multi-turn conversations - // This is especially important when OpenAI SDKs don't preserve extra_content - redisClient - .setex( - `thought_signature:${toolCall.id}`, - 86400, // 1 day expiration - part.thoughtSignature, - ) - .catch((err) => { - logger.error( - "Failed to cache thought_signature in streaming", - { - err, - }, - ); - }); } return toolCall; }) || null diff --git a/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts b/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts index 9cb422b70b..0a86278a6c 100644 --- a/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts +++ b/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts @@ -1,3 +1,4 @@ +import { redisClient } from "@llmgateway/cache"; import { logger } from "@llmgateway/logger"; import { calculatePromptTokensFromMessages } from "./calculate-prompt-tokens.js"; @@ -423,8 +424,10 @@ export function transformStreamingToOpenai( if (part.functionCall) { const callIndex = toolCalls.length; + const toolCallId = + part.functionCall.name + "_" + Date.now() + "_" + callIndex; toolCalls.push({ - id: part.functionCall.name + "_" + Date.now() + "_" + callIndex, + id: toolCallId, type: "function", index: partIndex, function: { @@ -443,6 +446,23 @@ export function transformStreamingToOpenai( } : undefined, }); + + // Cache thoughtSignature in Redis for server-side retrieval in multi-turn conversations + // This is especially important when OpenAI SDKs don't preserve extra_content/provider_extra + if (sig) { + redisClient + .setex( + `thought_signature:${toolCallId}`, + 86400, // 1 day expiration + sig, + ) + .catch((err) => { + logger.error( + "Failed to cache thought_signature in streaming transform", + { err }, + ); + }); + } } if (sig) {