Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/fixes/7821-gpt56-codex-responses-lite.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- fix(sse): preserve parallel_tool_calls for GPT-5.6 ultra/max delegation under Codex Responses Lite (#7821)
23 changes: 20 additions & 3 deletions open-sse/executors/codex.ts
Original file line number Diff line number Diff line change
Expand Up @@ -153,15 +153,28 @@ function isCodexResponsesLiteRequest(
);
}

// GPT-5.6 ultra-tier (sol/terra at "ultra") and luna at "max" coordinate delegation to
// sub-agents via parallel tool calls (see the effort-clamp comment near clampEffort()).
// Responses Lite must not strip parallel_tool_calls for those model/effort combos, or
// delegation silently breaks while the request still returns HTTP 200 (issue #7821).
function isCodexDelegationDependentModel(model: unknown): boolean {
const { baseModel, effort } = splitCodexReasoningSuffix(model);
if (effort === "ultra" && GPT_5_6_ULTRA_ALIAS_MODELS.has(baseModel)) return true;
if (effort === "max" && baseModel === "gpt-5.6-luna") return true;
return false;
}

function enforceCodexResponsesLiteParallelToolCalls(
bodyInput: unknown,
clientHeaders?: Record<string, string> | null
clientHeaders: Record<string, string> | null | undefined,
model: unknown
): unknown {
if (
!isCodexResponsesLiteRequest(bodyInput, clientHeaders) ||
!bodyInput ||
typeof bodyInput !== "object" ||
Array.isArray(bodyInput)
Array.isArray(bodyInput) ||
isCodexDelegationDependentModel(model)
) {
return bodyInput;
}
Comment on lines +160 to 180

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

The current implementation of isCodexDelegationDependentModel only checks the model suffix (e.g., gpt-5.6-sol-ultra) to determine if the request is delegation-dependent. However, clients can also specify the reasoning effort in the request body via reasoning.effort or reasoning_effort (e.g., with a base model gpt-5.6-sol and reasoning: { effort: "ultra" }). In such cases, isCodexDelegationDependentModel will return false, and enforceCodexResponsesLiteParallelToolCalls will incorrectly force parallel_tool_calls: false, silently breaking delegation.

We should update isCodexDelegationDependentModel to also inspect the request body for the reasoning effort if it is not present in the model suffix.

function isCodexDelegationDependentModel(model: unknown, bodyInput: unknown): boolean {
  const { baseModel, effort: modelEffort } = splitCodexReasoningSuffix(model);
  let effort = modelEffort;
  if (!effort && bodyInput && typeof bodyInput === "object" && !Array.isArray(bodyInput)) {
    const body = bodyInput as Record<string, unknown>;
    const reasoningRecord =
      body.reasoning && typeof body.reasoning === "object" && !Array.isArray(body.reasoning)
        ? (body.reasoning as Record<string, unknown>)
        : null;
    const explicitReasoning = typeof reasoningRecord?.effort === "string" ? reasoningRecord.effort.trim().toLowerCase() : undefined;
    const requestReasoningEffort = typeof body.reasoning_effort === "string" ? body.reasoning_effort.trim().toLowerCase() : undefined;
    effort = (explicitReasoning || requestReasoningEffort || null) as EffortLevel | null;
  }
  if (effort === "ultra" && GPT_5_6_ULTRA_ALIAS_MODELS.has(baseModel)) return true;
  if (effort === "max" && baseModel === "gpt-5.6-luna") return true;
  return false;
}

function enforceCodexResponsesLiteParallelToolCalls(
  bodyInput: unknown,
  clientHeaders: Record<string, string> | null | undefined,
  model: unknown
): unknown {
  if (
    !isCodexResponsesLiteRequest(bodyInput, clientHeaders) ||
    !bodyInput ||
    typeof bodyInput !== "object" ||
    Array.isArray(bodyInput) ||
    isCodexDelegationDependentModel(model, bodyInput)
  ) {
    return bodyInput;
  }

Expand Down Expand Up @@ -847,7 +860,11 @@ export class CodexExecutor extends BaseExecutor {
}

async execute(input: ExecuteInput) {
const requestBody = enforceCodexResponsesLiteParallelToolCalls(input.body, input.clientHeaders);
const requestBody = enforceCodexResponsesLiteParallelToolCalls(
input.body,
input.clientHeaders,
input.model
);
const requestInput = requestBody === input.body ? input : { ...input, body: requestBody };
const sessionId = this.getPromptCacheSessionId(
requestInput.credentials,
Expand Down
89 changes: 89 additions & 0 deletions tests/unit/executor-codex-gpt56-lite-ultra.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,89 @@
import test from "node:test";
import assert from "node:assert/strict";

import { CodexExecutor } from "../../open-sse/executors/codex.ts";

// Issue #7821: enforceCodexResponsesLiteParallelToolCalls() used to have zero model
// awareness — it force-set parallel_tool_calls:false for EVERY model when Responses
// Lite was detected, including gpt-5.6-sol/-terra at "ultra" effort (and gpt-5.6-luna
// at "max"), whose delegation-to-sub-agents capability depends on parallel_tool_calls
// staying enabled. This collided with the effort-clamp comment near clampEffort()
// ("Ultra coordinates delegation in Codex clients") and is why GPT-5.6 was reported
// unusable through the stock Codex CLI/App (which enables Responses Lite by default)
// while GPT-5.5 (no delegation tier) was unaffected.
//
// Covers the interaction (lite marker + delegation-dependent model/effort together),
// not just each behavior in isolation — that interaction was the actual blind spot.

async function runLiteRequest(model: string): Promise<Record<string, unknown>[]> {
const executor = new CodexExecutor();
const originalFetch = globalThis.fetch;
const capturedBodies: Record<string, unknown>[] = [];

globalThis.fetch = async (_url, init) => {
capturedBodies.push(JSON.parse(String(init?.body || "{}")));
return new Response(JSON.stringify({ id: "resp_lite", object: "response" }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};

const body = {
_nativeCodexPassthrough: true,
model,
input: [],
parallel_tool_calls: true,
};

try {
await executor.execute({
model,
body,
stream: true,
credentials: { accessToken: "codex-token" },
clientHeaders: { "X-OpenAI-Internal-Codex-Responses-Lite": "true" },
});
} finally {
globalThis.fetch = originalFetch;
}

return capturedBodies;
}
Comment on lines +18 to +51

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

To support testing delegation-dependent models where the effort is specified in the request body rather than the model suffix, we should update the runLiteRequest helper to accept an optional bodyOverride parameter.

async function runLiteRequest(model: string, bodyOverride?: Record<string, unknown>): Promise<Record<string, unknown>[]> {
  const executor = new CodexExecutor();
  const originalFetch = globalThis.fetch;
  const capturedBodies: Record<string, unknown>[] = [];

  globalThis.fetch = async (_url, init) => {
    capturedBodies.push(JSON.parse(String(init?.body || "{}")));
    return new Response(JSON.stringify({ id: "resp_lite", object: "response" }), {
      status: 200,
      headers: { "Content-Type": "application/json" },
    });
  };

  const body = {
    _nativeCodexPassthrough: true,
    model,
    input: [],
    parallel_tool_calls: true,
    ...bodyOverride,
  };

  try {
    await executor.execute({
      model,
      body,
      stream: true,
      credentials: { accessToken: "codex-token" },
      clientHeaders: { "X-OpenAI-Internal-Codex-Responses-Lite": "true" },
    });
  } finally {
    globalThis.fetch = originalFetch;
  }

  return capturedBodies;
}


test("Responses Lite must not strip parallel_tool_calls for GPT-5.6 sol ultra-tier delegation", async () => {
const capturedBodies = await runLiteRequest("gpt-5.6-sol-ultra");
assert.equal(
capturedBodies[0].parallel_tool_calls,
true,
"Responses Lite stripped parallel_tool_calls for an ultra-tier GPT-5.6 delegation " +
"request — this breaks sub-agent delegation and is the root cause of #7821"
);
});

test("Responses Lite must not strip parallel_tool_calls for GPT-5.6 terra ultra-tier delegation", async () => {
const capturedBodies = await runLiteRequest("gpt-5.6-terra-ultra");
assert.equal(capturedBodies[0].parallel_tool_calls, true);
});

test("Responses Lite must not strip parallel_tool_calls for GPT-5.6 luna max-tier delegation", async () => {
const capturedBodies = await runLiteRequest("gpt-5.6-luna-max");
assert.equal(capturedBodies[0].parallel_tool_calls, true);
});

test("Responses Lite still forces parallel_tool_calls:false for non-delegation GPT-5.5", async () => {
const capturedBodies = await runLiteRequest("gpt-5.5");
assert.equal(
capturedBodies[0].parallel_tool_calls,
false,
"GPT-5.5 has no delegation tier — Responses Lite behavior for it must be unchanged"
);
});

test("Responses Lite still forces parallel_tool_calls:false for GPT-5.6 sol at non-ultra effort", async () => {
const capturedBodies = await runLiteRequest("gpt-5.6-sol-high");
assert.equal(
capturedBodies[0].parallel_tool_calls,
false,
"Non-ultra GPT-5.6 effort tiers have no delegation dependency — must stay forced off"
);
});
Comment on lines +82 to +89

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

Add regression tests to verify that parallel_tool_calls is preserved when the delegation-dependent effort (ultra or max) is specified in the request body (either via reasoning.effort or reasoning_effort) instead of the model suffix.

test(

Loading