Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 33 additions & 1 deletion open-sse/utils/streamReadinessPolicy.ts
Original file line number Diff line number Diff line change
Expand Up @@ -95,6 +95,27 @@ function isHighReasoningEffort(
return effort.toLowerCase() === "high";
}

/**
* Extended-thinking aliases (`claude-sonnet-5-thinking`, `claude-opus-4-6-thinking-1m`,
* `gpt-5.6-luna-free-thinking`, ...) run the same cold reasoning warm-up that earns
* codex high-effort targets their unconditional bump — the client asked for extended
* thinking, so the first non-ping event is gated behind it regardless of prompt size.
*
* This is keyed on the model alias rather than the provider's registry `format`
* because the providers that serve these models translate them themselves:
* kiro (`format: "kiro"`, Anthropic models over CodeWhisperer), devin, antigravity
* and chatgpt-web all publish `-thinking` ids while sitting outside the
* claude-format bump. #11922 was exactly that gap — `kiro/claude-sonnet-5-thinking`
* 504'd at the 125s readiness window (80s base + 45s large-history) because nothing
* in this policy recognised the request as a reasoning target.
*/
function isExtendedThinkingModel(model?: string | null): boolean {
if (!model) return false;
// Matches `-thinking` at the end and before a further qualifier (`-thinking-1m`),
// without matching unrelated ids that merely contain the word.
return /-thinking(?:-|$)/.test(model.toLowerCase());
}

export function resolveStreamReadinessTimeout(
input: StreamReadinessPolicyInput
): StreamReadinessPolicyResult {
Expand All @@ -114,6 +135,7 @@ export function resolveStreamReadinessTimeout(
const estimatedChars = estimateBodyChars(input.body);
const codexGpt5x = isCodexGpt5x(input.provider, input.model);
const codexHighReasoning = codexGpt5x && isHighReasoningEffort(input.model, input.body);
const extendedThinking = isExtendedThinkingModel(input.model);

if (itemCount > VERY_LARGE_ITEM_THRESHOLD) {
timeoutMs += 45_000;
Expand Down Expand Up @@ -157,7 +179,17 @@ export function resolveStreamReadinessTimeout(
// first SSE event — enough that the default 80s readiness window 504s before
// the upstream speaks. Mirror the codex_gpt_5_5_high_reasoning bump so this
// class of provider cannot be misidentified as a stalled connection.
if (isClaudeFormatReasoningProvider(input.provider) && !codexHighReasoning) {
// #11922: an explicitly-requested extended-thinking target warms up before it
// emits anything, on whichever provider serves it. Mirror the codex high-effort
// bump so the readiness watchdog cannot mistake that warm-up for a stalled
// stream. Kept exclusive with the other reasoning bumps below — these all model
// the same one-off warm-up, so they must not stack into a multi-minute window.
if (extendedThinking && !codexHighReasoning) {
timeoutMs += 30_000;
reasons.push("extended_thinking");
}

if (isClaudeFormatReasoningProvider(input.provider) && !codexHighReasoning && !extendedThinking) {
timeoutMs += 30_000;
reasons.push("claude_format_heavy_reasoning");
}
Expand Down
73 changes: 72 additions & 1 deletion tests/unit/stream-readiness-policy.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -163,7 +163,7 @@ test("bumps small requests to third-party Claude-format replicas (agentrouter, Z
);
});

test("does NOT bump Minimax (M3) — #3110 moved it from claude to openai format so images work, and the readiness bump is keyed off the registry's `format: \"claude\"` field", () => {
test('does NOT bump Minimax (M3) — #3110 moved it from claude to openai format so images work, and the readiness bump is keyed off the registry\'s `format: "claude"` field', () => {
// Minimax's replica quirk (long reasoning warm-up) hasn't changed, but this
// policy intentionally keys off the translator format, not the provider
// name — the registry is the single source of truth (see isClaudeFormatReasoningProvider
Expand Down Expand Up @@ -266,3 +266,74 @@ test("treats unknown provider names as non-Claude-format (no false positives)",
assert.equal(result.timeoutMs, 80_000);
assert.ok(!result.reasons.includes("claude_format_heavy_reasoning"));
});

test("gives extended-thinking model aliases the reasoning readiness bump (#11922)", () => {
// #11922: kiro/claude-sonnet-5-thinking 504'd with
// "Stream produced no non-ping SSE event within 125000ms" — the 80s base plus
// the 45s very-large-history bump, capped there because nothing recognised the
// request as a reasoning target. Kiro serves Anthropic thinking models through
// its own CodeWhisperer translator, so `format: "kiro"` (not "claude") kept it
// out of the claude_format_heavy_reasoning bump, and the `-thinking` alias was
// never a reasoning signal the way `-high` is.
const result = resolveStreamReadinessTimeout({
baseTimeoutMs: 80_000,
provider: "kiro",
model: "claude-sonnet-5-thinking",
body: { messages: items(401) },
});

assert.equal(result.timeoutMs, 155_000);
assert.ok(result.reasons.includes("extended_thinking"));
});

test("extended-thinking bump is provider-agnostic and fires for small requests", () => {
const result = resolveStreamReadinessTimeout({
baseTimeoutMs: 80_000,
provider: "devin",
model: "claude-opus-4-6-thinking",
body: { messages: items(2) },
});

assert.equal(result.timeoutMs, 110_000);
assert.ok(result.reasons.includes("extended_thinking"));
});

test("does NOT stack extended-thinking with the Claude-format replica bump", () => {
// Both bumps model the same one-off reasoning warm-up. A claude-format replica
// serving a `-thinking` alias must get 30s once, not 60s twice.
const result = resolveStreamReadinessTimeout({
baseTimeoutMs: 80_000,
provider: "agentrouter",
model: "claude-sonnet-4-6-thinking",
body: { messages: items(2) },
});

assert.equal(result.timeoutMs, 110_000);
assert.ok(result.reasons.includes("extended_thinking"));
assert.ok(!result.reasons.includes("claude_format_heavy_reasoning"));
});

test("does NOT stack extended-thinking with the codex high-reasoning bump", () => {
const result = resolveStreamReadinessTimeout({
baseTimeoutMs: 80_000,
provider: "codex",
model: "gpt-5.5-thinking",
body: { messages: items(2), reasoning_effort: "high" },
});

assert.equal(result.timeoutMs, 110_000);
assert.ok(result.reasons.includes("codex_gpt_5_5_high_reasoning"));
assert.ok(!result.reasons.includes("extended_thinking"));
});

test("does not treat an unrelated id containing 'thinking' as an alias suffix", () => {
const result = resolveStreamReadinessTimeout({
baseTimeoutMs: 80_000,
provider: "openai",
model: "thinking-machines-lab-model",
body: { messages: items(2) },
});

assert.equal(result.timeoutMs, 80_000);
assert.ok(!result.reasons.includes("extended_thinking"));
});
Loading