From f1055bd070e23f8914c806809316288582362286 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Tue, 23 Jun 2026 18:09:47 -0300 Subject: [PATCH] fix(sse): strip reasoning blobs from agentic context to prevent O(n^2) token growth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reasoning summaries were accumulating across agentic turns, inflating the prompt on every subsequent request: - Codex Responses `stripStoredItemReferences` now drops object items with `type:"reasoning"`. Their `encrypted_content` is unusable with `store=false` (the `previous_response_id` is deleted), so they only waste context tokens. - `filterToOpenAIFormat` strips `reasoning_content` from assistant messages that carry `tool_calls`, instead of returning them as-is. The Responses-input buffering machinery (pendingReasoning / extractReasoningText) that upstream removes is already absent in OmniRoute — reasoning items are skipped there. Co-authored-by: GodrezJr2 Inspired-by: https://github.com/decolua/9router/pull/1599 --- open-sse/executors/codex.ts | 15 +++- open-sse/translator/helpers/openaiHelper.ts | 11 ++- tests/unit/executor-codex.test.ts | 5 +- ...asoning-blobs-agentic-context-1599.test.ts | 81 +++++++++++++++++++ 4 files changed, 108 insertions(+), 4 deletions(-) create mode 100644 tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index ffa7b0d81df..172fa1a9874 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -296,7 +296,7 @@ function convertSystemToDeveloperRole(body: Record): void { * server-generated prefix (rs_, fc_, resp_, msg_) — so the content is * preserved but the backend won't try to look it up */ -function stripStoredItemReferences(body: Record): void { +export function stripStoredItemReferences(body: Record): void { if (Array.isArray(body.input) && body.input.length === 0) { body.input = [ { @@ -330,6 +330,19 @@ function stripStoredItemReferences(body: Record): void { return false; } + // Reasoning blobs (encrypted_content) are unusable with store=false since + // previous_response_id is deleted — strip them to avoid wasting context + // tokens (O(n^2) growth across agentic turns). + if ( + item && + typeof item === "object" && + !Array.isArray(item) && + (item as Record).type === "reasoning" + ) { + strippedCount++; + return false; + } + // Object items with server-generated IDs: strip the id field but keep the item. // e.g. { id: "rs_...", type: "reasoning", summary: [...] } → keep content, remove id // e.g. { id: "fc_...", type: "function_call", ... } → keep content, remove id diff --git a/open-sse/translator/helpers/openaiHelper.ts b/open-sse/translator/helpers/openaiHelper.ts index 67186eaa69f..b6a9bbfb321 100644 --- a/open-sse/translator/helpers/openaiHelper.ts +++ b/open-sse/translator/helpers/openaiHelper.ts @@ -43,8 +43,15 @@ export function filterToOpenAIFormat(body) { // Keep tool messages as-is (OpenAI format) if (msg.role === "tool") return msg; - // Keep assistant messages with tool_calls as-is - if (msg.role === "assistant" && msg.tool_calls) return msg; + // Keep assistant messages with tool_calls, but strip reasoning_content — + // reasoning blobs inflate context on every subsequent agentic turn (O(n^2)). + if (msg.role === "assistant" && msg.tool_calls) { + if (msg.reasoning_content !== undefined) { + const { reasoning_content, ...cleanMsg } = msg; + return cleanMsg; + } + return msg; + } // Handle string content if (typeof msg.content === "string") return msg; diff --git a/tests/unit/executor-codex.test.ts b/tests/unit/executor-codex.test.ts index 3ece99efe36..7fb6e3be523 100644 --- a/tests/unit/executor-codex.test.ts +++ b/tests/unit/executor-codex.test.ts @@ -530,9 +530,12 @@ test("CodexExecutor.transformRequest preserves native assistant commentary histo ), true ); + // Reasoning items are stripped from the Responses input — encrypted_content is + // unusable with store=false (previous_response_id deleted) and the summary blob + // only inflates context on every subsequent agentic turn (decolua/9router#1599). assert.equal( result.input.some((item) => item.type === "reasoning"), - true + false ); assert.equal( result.input.some((item) => item.type === "function_call"), diff --git a/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts new file mode 100644 index 00000000000..f19a12f5f62 --- /dev/null +++ b/tests/unit/strip-reasoning-blobs-agentic-context-1599.test.ts @@ -0,0 +1,81 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { stripStoredItemReferences } from "../../open-sse/executors/codex.ts"; +import { filterToOpenAIFormat } from "../../open-sse/translator/helpers/openaiHelper.ts"; + +// Port of decolua/9router#1599 — strip reasoning blobs from agentic context to +// prevent O(n^2) token growth across turns. +// +// (1) codex.ts stripStoredItemReferences: object items of type "reasoning" +// (encrypted_content) are unusable with store=false (previous_response_id is +// deleted) and must be dropped from the Responses `input` array. +// (2) openaiHelper.ts filterToOpenAIFormat: assistant+tool_calls messages must +// have `reasoning_content` stripped instead of being returned as-is. + +test("stripStoredItemReferences drops object items with type=reasoning", () => { + const body: Record = { + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { id: "rs_abc123", type: "reasoning", summary: [{ text: "thinking..." }] }, + { type: "reasoning", encrypted_content: "blob" }, + { + type: "function_call", + id: "fc_xyz789", + name: "search", + arguments: "{}", + call_id: "call_1", + }, + ], + }; + + stripStoredItemReferences(body); + + const input = body.input as Array>; + // Both reasoning items must be gone. + assert.equal( + input.some((it) => it && it.type === "reasoning"), + false, + "reasoning items must be stripped" + ); + // Non-reasoning items survive (message + function_call), id is sanitized. + assert.equal(input.length, 2); + assert.equal(input[0].type, "message"); + assert.equal(input[1].type, "function_call"); + assert.equal(input[1].id, undefined, "fc_ server id stripped, item kept"); +}); + +test("filterToOpenAIFormat strips reasoning_content from assistant+tool_calls messages", () => { + const body = { + messages: [ + { + role: "assistant", + reasoning_content: "long chain of thought that inflates context", + tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: "{}" } }], + }, + ], + }; + + const result = filterToOpenAIFormat(body) as { messages: Array> }; + const msg = result.messages[0]; + + assert.equal(msg.reasoning_content, undefined, "reasoning_content must be dropped"); + assert.ok(Array.isArray(msg.tool_calls), "tool_calls preserved"); + assert.equal((msg.tool_calls as unknown[]).length, 1); + assert.equal(msg.role, "assistant"); +}); + +test("filterToOpenAIFormat keeps assistant+tool_calls untouched when no reasoning_content", () => { + const body = { + messages: [ + { + role: "assistant", + content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: "{}" } }], + }, + ], + }; + + const result = filterToOpenAIFormat(body) as { messages: Array> }; + assert.deepEqual(result.messages[0], body.messages[0]); +});