diff --git a/apps/gateway/src/chat/tools/extract-tool-calls.ts b/apps/gateway/src/chat/tools/extract-tool-calls.ts index dadf70e106..b941a0c138 100644 --- a/apps/gateway/src/chat/tools/extract-tool-calls.ts +++ b/apps/gateway/src/chat/tools/extract-tool-calls.ts @@ -1,6 +1,3 @@ -import { redisClient } from "@llmgateway/cache"; -import { logger } from "@llmgateway/logger"; - import type { Provider } from "@llmgateway/models"; /** @@ -45,6 +42,8 @@ export function extractToolCalls(data: any, provider: Provider): any[] | null { case "google-vertex": { // Google AI Studio tool calls in streaming // Include thoughtSignature if present (required for Gemini 3 multi-turn conversations) + // Note: Redis caching of thought_signature happens in transform-streaming-to-openai.ts + // where the actual tool_call ID sent to clients is generated const parts = data.candidates?.[0]?.content?.parts || []; return ( parts @@ -65,22 +64,6 @@ export function extractToolCalls(data: any, provider: Provider): any[] | null { thought_signature: part.thoughtSignature, }, }; - // Cache thoughtSignature in Redis for server-side retrieval in multi-turn conversations - // This is especially important when OpenAI SDKs don't preserve extra_content - redisClient - .setex( - `thought_signature:${toolCall.id}`, - 86400, // 1 day expiration - part.thoughtSignature, - ) - .catch((err) => { - logger.error( - "Failed to cache thought_signature in streaming", - { - err, - }, - ); - }); } return toolCall; }) || null diff --git a/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts b/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts index 9cb422b70b..0a86278a6c 100644 --- a/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts +++ b/apps/gateway/src/chat/tools/transform-streaming-to-openai.ts @@ -1,3 +1,4 @@ +import { redisClient } from "@llmgateway/cache"; import { logger } from "@llmgateway/logger"; import { calculatePromptTokensFromMessages } from "./calculate-prompt-tokens.js"; @@ -423,8 +424,10 @@ export function transformStreamingToOpenai( if (part.functionCall) { const callIndex = toolCalls.length; + const toolCallId = + part.functionCall.name + "_" + Date.now() + "_" + callIndex; toolCalls.push({ - id: part.functionCall.name + "_" + Date.now() + "_" + callIndex, + id: toolCallId, type: "function", index: partIndex, function: { @@ -443,6 +446,23 @@ export function transformStreamingToOpenai( } : undefined, }); + + // Cache thoughtSignature in Redis for server-side retrieval in multi-turn conversations + // This is especially important when OpenAI SDKs don't preserve extra_content/provider_extra + if (sig) { + redisClient + .setex( + `thought_signature:${toolCallId}`, + 86400, // 1 day expiration + sig, + ) + .catch((err) => { + logger.error( + "Failed to cache thought_signature in streaming transform", + { err }, + ); + }); + } } if (sig) {