diff --git a/open-sse/utils/streamReadinessPolicy.ts b/open-sse/utils/streamReadinessPolicy.ts index 2dcc3748090..0507317bc31 100644 --- a/open-sse/utils/streamReadinessPolicy.ts +++ b/open-sse/utils/streamReadinessPolicy.ts @@ -95,6 +95,27 @@ function isHighReasoningEffort( return effort.toLowerCase() === "high"; } +/** + * Extended-thinking aliases (`claude-sonnet-5-thinking`, `claude-opus-4-6-thinking-1m`, + * `gpt-5.6-luna-free-thinking`, ...) run the same cold reasoning warm-up that earns + * codex high-effort targets their unconditional bump — the client asked for extended + * thinking, so the first non-ping event is gated behind it regardless of prompt size. + * + * This is keyed on the model alias rather than the provider's registry `format` + * because the providers that serve these models translate them themselves: + * kiro (`format: "kiro"`, Anthropic models over CodeWhisperer), devin, antigravity + * and chatgpt-web all publish `-thinking` ids while sitting outside the + * claude-format bump. #11922 was exactly that gap — `kiro/claude-sonnet-5-thinking` + * 504'd at the 125s readiness window (80s base + 45s large-history) because nothing + * in this policy recognised the request as a reasoning target. + */ +function isExtendedThinkingModel(model?: string | null): boolean { + if (!model) return false; + // Matches `-thinking` at the end and before a further qualifier (`-thinking-1m`), + // without matching unrelated ids that merely contain the word. + return /-thinking(?:-|$)/.test(model.toLowerCase()); +} + export function resolveStreamReadinessTimeout( input: StreamReadinessPolicyInput ): StreamReadinessPolicyResult { @@ -114,6 +135,7 @@ export function resolveStreamReadinessTimeout( const estimatedChars = estimateBodyChars(input.body); const codexGpt5x = isCodexGpt5x(input.provider, input.model); const codexHighReasoning = codexGpt5x && isHighReasoningEffort(input.model, input.body); + const extendedThinking = isExtendedThinkingModel(input.model); if (itemCount > VERY_LARGE_ITEM_THRESHOLD) { timeoutMs += 45_000; @@ -157,7 +179,17 @@ export function resolveStreamReadinessTimeout( // first SSE event — enough that the default 80s readiness window 504s before // the upstream speaks. Mirror the codex_gpt_5_5_high_reasoning bump so this // class of provider cannot be misidentified as a stalled connection. - if (isClaudeFormatReasoningProvider(input.provider) && !codexHighReasoning) { + // #11922: an explicitly-requested extended-thinking target warms up before it + // emits anything, on whichever provider serves it. Mirror the codex high-effort + // bump so the readiness watchdog cannot mistake that warm-up for a stalled + // stream. Kept exclusive with the other reasoning bumps below — these all model + // the same one-off warm-up, so they must not stack into a multi-minute window. + if (extendedThinking && !codexHighReasoning) { + timeoutMs += 30_000; + reasons.push("extended_thinking"); + } + + if (isClaudeFormatReasoningProvider(input.provider) && !codexHighReasoning && !extendedThinking) { timeoutMs += 30_000; reasons.push("claude_format_heavy_reasoning"); } diff --git a/tests/unit/stream-readiness-policy.test.ts b/tests/unit/stream-readiness-policy.test.ts index 0292f24b1df..2ac772d297e 100644 --- a/tests/unit/stream-readiness-policy.test.ts +++ b/tests/unit/stream-readiness-policy.test.ts @@ -163,7 +163,7 @@ test("bumps small requests to third-party Claude-format replicas (agentrouter, Z ); }); -test("does NOT bump Minimax (M3) — #3110 moved it from claude to openai format so images work, and the readiness bump is keyed off the registry's `format: \"claude\"` field", () => { +test('does NOT bump Minimax (M3) — #3110 moved it from claude to openai format so images work, and the readiness bump is keyed off the registry\'s `format: "claude"` field', () => { // Minimax's replica quirk (long reasoning warm-up) hasn't changed, but this // policy intentionally keys off the translator format, not the provider // name — the registry is the single source of truth (see isClaudeFormatReasoningProvider @@ -266,3 +266,74 @@ test("treats unknown provider names as non-Claude-format (no false positives)", assert.equal(result.timeoutMs, 80_000); assert.ok(!result.reasons.includes("claude_format_heavy_reasoning")); }); + +test("gives extended-thinking model aliases the reasoning readiness bump (#11922)", () => { + // #11922: kiro/claude-sonnet-5-thinking 504'd with + // "Stream produced no non-ping SSE event within 125000ms" — the 80s base plus + // the 45s very-large-history bump, capped there because nothing recognised the + // request as a reasoning target. Kiro serves Anthropic thinking models through + // its own CodeWhisperer translator, so `format: "kiro"` (not "claude") kept it + // out of the claude_format_heavy_reasoning bump, and the `-thinking` alias was + // never a reasoning signal the way `-high` is. + const result = resolveStreamReadinessTimeout({ + baseTimeoutMs: 80_000, + provider: "kiro", + model: "claude-sonnet-5-thinking", + body: { messages: items(401) }, + }); + + assert.equal(result.timeoutMs, 155_000); + assert.ok(result.reasons.includes("extended_thinking")); +}); + +test("extended-thinking bump is provider-agnostic and fires for small requests", () => { + const result = resolveStreamReadinessTimeout({ + baseTimeoutMs: 80_000, + provider: "devin", + model: "claude-opus-4-6-thinking", + body: { messages: items(2) }, + }); + + assert.equal(result.timeoutMs, 110_000); + assert.ok(result.reasons.includes("extended_thinking")); +}); + +test("does NOT stack extended-thinking with the Claude-format replica bump", () => { + // Both bumps model the same one-off reasoning warm-up. A claude-format replica + // serving a `-thinking` alias must get 30s once, not 60s twice. + const result = resolveStreamReadinessTimeout({ + baseTimeoutMs: 80_000, + provider: "agentrouter", + model: "claude-sonnet-4-6-thinking", + body: { messages: items(2) }, + }); + + assert.equal(result.timeoutMs, 110_000); + assert.ok(result.reasons.includes("extended_thinking")); + assert.ok(!result.reasons.includes("claude_format_heavy_reasoning")); +}); + +test("does NOT stack extended-thinking with the codex high-reasoning bump", () => { + const result = resolveStreamReadinessTimeout({ + baseTimeoutMs: 80_000, + provider: "codex", + model: "gpt-5.5-thinking", + body: { messages: items(2), reasoning_effort: "high" }, + }); + + assert.equal(result.timeoutMs, 110_000); + assert.ok(result.reasons.includes("codex_gpt_5_5_high_reasoning")); + assert.ok(!result.reasons.includes("extended_thinking")); +}); + +test("does not treat an unrelated id containing 'thinking' as an alias suffix", () => { + const result = resolveStreamReadinessTimeout({ + baseTimeoutMs: 80_000, + provider: "openai", + model: "thinking-machines-lab-model", + body: { messages: items(2) }, + }); + + assert.equal(result.timeoutMs, 80_000); + assert.ok(!result.reasons.includes("extended_thinking")); +});