diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index 6d70772a6f0..df275ba0219 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -895,6 +895,20 @@ export async function validateResponseQuality( } if (!hasContent && !hasToolCalls) { + // finish_reason "length" is a truncated completion (max_tokens hit), the + // same case the Claude shape exempts as stop_reason "max_tokens" (#12968). + // A thinking model that spends the whole budget before any visible token + // is a valid response, not a reason to fail the combo target over. + if (firstChoice?.finish_reason === "length") { + return { + valid: true, + clonedResponse: new Response(text, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }), + }; + } return { valid: false, reason: "empty content and no tool_calls in response" }; } diff --git a/open-sse/utils/diagnostics.ts b/open-sse/utils/diagnostics.ts index 82034ac9af5..2d569d1cc82 100644 --- a/open-sse/utils/diagnostics.ts +++ b/open-sse/utils/diagnostics.ts @@ -345,7 +345,21 @@ export function detectMalformedNonStream( return false; }); - if (!anyHasOutput) return "empty_choices"; + if (!anyHasOutput) { + // A finish_reason of "length" is the chat-completions spelling of a + // truncated completion: the model hit max_tokens. Claude's translator maps + // stop_reason "max_tokens" to it (claude-to-openai.ts), and the Claude + // shape already exempts that case (#12968, diagnostics above) because a + // thinking model can burn a 1-token probe budget and return no visible + // text. Rejecting the translated form reintroduces the 502 the exemption + // removed. "stop" with no output stays empty_choices. + const truncated = choices.some((choice) => { + const c = choice as Record; + return c?.finish_reason === "length"; + }); + if (truncated) return null; + return "empty_choices"; + } // #13461: only for the narrow provider allowlist — see classifyFakeSuccessBody's // doc comment for the false-positive guards (short content + dominant signal). diff --git a/tests/unit/diagnostics.test.ts b/tests/unit/diagnostics.test.ts index d5b5d1dcff6..7733b99813c 100644 --- a/tests/unit/diagnostics.test.ts +++ b/tests/unit/diagnostics.test.ts @@ -69,6 +69,34 @@ test("detectMalformedNonStream allows Claude message with (empty response) + sto ); }); +// Chat-completions spelling of the same truncation. Claude's translator maps +// stop_reason "max_tokens" to finish_reason "length"; a thinking model can +// burn a 1-token probe budget and return no visible text. The Claude shape +// exempts that (#12968); the translated shape must too. +test("detectMalformedNonStream allows a chat completion truncated at length with no visible text", () => { + const resp = { + choices: [{ finish_reason: "length", message: { role: "assistant", content: null } }], + }; + assert.equal(detectMalformedNonStream(resp), null); +}); + +test("detectMalformedNonStream still rejects a chat completion that stopped with no output", () => { + const resp = { + choices: [{ finish_reason: "stop", message: { role: "assistant", content: null } }], + }; + assert.equal(detectMalformedNonStream(resp), "empty_choices"); +}); + +// finishReason.ts normalizes "max_tokens" to "length" before this function +// sees it. If a caller bypasses that normalization, the raw "max_tokens" +// spelling must still be rejected — only the normalized "length" is exempt. +test("detectMalformedNonStream rejects a chat completion with raw max_tokens and no output", () => { + const resp = { + choices: [{ finish_reason: "max_tokens", message: { role: "assistant", content: null } }], + }; + assert.equal(detectMalformedNonStream(resp), "empty_choices"); +}); + // ── (b) synthResponsesFailure matches a response.failed event ──────────────── test("synthResponsesFailure produces a response.failed SSE event", () => { diff --git a/tests/unit/validate-response-quality.test.ts b/tests/unit/validate-response-quality.test.ts index 6a918155e56..a134840267c 100644 --- a/tests/unit/validate-response-quality.test.ts +++ b/tests/unit/validate-response-quality.test.ts @@ -166,3 +166,31 @@ test("streaming OpenAI finish_reason-only chunk (no content delta) → invalid ( assert.strictEqual(verdict.valid, false); assert.match(verdict.reason ?? "", /streaming openai terminated with empty completion/); }); + +function chatBody(finishReason: string) { + return JSON.stringify({ + choices: [{ finish_reason: finishReason, message: { role: "assistant", content: null } }], + }); +} + +// finish_reason "length" is the chat spelling of a max_tokens truncation, the +// case the Claude shape already exempts (#12968). A thinking model can spend +// the whole budget before emitting text; that must not fail the combo over. +test("non-streaming chat completion truncated at length with no text is valid", async () => { + const verdict = await validateResponseQuality( + makeResponse(chatBody("length"), "application/json"), + false, + {} + ); + assert.strictEqual(verdict.valid, true); +}); + +test("non-streaming chat completion that stopped with no text stays invalid", async () => { + const verdict = await validateResponseQuality( + makeResponse(chatBody("stop"), "application/json"), + false, + {} + ); + assert.strictEqual(verdict.valid, false); + assert.match(verdict.reason ?? "", /empty content/); +});