diff --git a/apps/docs/content/features/reasoning.mdx b/apps/docs/content/features/reasoning.mdx index 985a0b7508..029773f6ff 100644 --- a/apps/docs/content/features/reasoning.mdx +++ b/apps/docs/content/features/reasoning.mdx @@ -40,7 +40,7 @@ Add the `reasoning_effort` parameter directly to your request: - `medium` - Balanced reasoning for most tasks - `high` - Deep reasoning for complex problems - `xhigh` - Very deep reasoning for the most complex problems -- `max` - Highest reasoning tier, above `xhigh`. Supported by Anthropic thinking models and OpenAI GPT-5.6 models. Effort values are forwarded to the provider as-is, so sending a value the target model doesn't support results in a provider error +- `max` - Highest reasoning tier, above `xhigh`. Supported by Anthropic thinking models and OpenAI GPT-5.6 models. Effort tiers are never downgraded by the gateway: providers that accept an effort parameter receive the value unchanged (unsupported values result in a provider error), while providers that take a thinking budget instead (Anthropic, Google) have each tier translated to a native budget OpenAI's reasoning models do not all accept the same effort values. The @@ -78,7 +78,7 @@ Use the unified `reasoning` configuration object with an `effort` field: - `medium` - Balanced reasoning for most tasks - `high` - Deep reasoning for complex problems - `xhigh` - Very deep reasoning for the most complex problems -- `max` - Highest reasoning tier, above `xhigh`. Supported by Anthropic thinking models and OpenAI GPT-5.6 models. Effort values are forwarded to the provider as-is, so sending a value the target model doesn't support results in a provider error +- `max` - Highest reasoning tier, above `xhigh`. Supported by Anthropic thinking models and OpenAI GPT-5.6 models. Effort tiers are never downgraded by the gateway: providers that accept an effort parameter receive the value unchanged (unsupported values result in a provider error), while providers that take a thinking budget instead (Anthropic, Google) have each tier translated to a native budget ```bash curl -X POST "https://api.llmgateway.io/v1/chat/completions" \ diff --git a/apps/gateway/src/anthropic/anthropic.ts b/apps/gateway/src/anthropic/anthropic.ts index 930c002ef0..b54cbfe5ed 100644 --- a/apps/gateway/src/anthropic/anthropic.ts +++ b/apps/gateway/src/anthropic/anthropic.ts @@ -215,7 +215,8 @@ const anthropicRequestSchema = z.object({ .object({ // Matches the chat completions reasoning-effort enum. Claude Code emits // the full range (including `xhigh` and `max`), so accept all of them; - // values are forwarded to the provider as-is. + // tiers are never downgraded — downstream they map onto Anthropic's + // native thinking controls (adaptive effort or a budget). effort: z .enum(["none", "minimal", "low", "medium", "high", "xhigh", "max"]) .optional(), diff --git a/apps/gateway/src/chat/schemas/completions.ts b/apps/gateway/src/chat/schemas/completions.ts index 35f216c72c..45c1c364be 100644 --- a/apps/gateway/src/chat/schemas/completions.ts +++ b/apps/gateway/src/chat/schemas/completions.ts @@ -307,7 +307,7 @@ export const completionsRequestSchema = z.object({ .transform((val) => (val === null ? undefined : val)) .openapi({ description: - "Controls the reasoning effort for reasoning-capable models. `none` is only supported by OpenAI's newer reasoning models (e.g. gpt-5.4 and later); for other providers it disables reasoning. `max` is the highest tier (above `xhigh`), supported by Anthropic models and OpenAI GPT-5.6 models. Values are forwarded to the provider as-is; sending a value the target model doesn't support results in a provider error. The exact values each provider mapping accepts are exposed as `reasoning_efforts` on `/v1/models`.", + "Controls the reasoning effort for reasoning-capable models. `none` is only supported by OpenAI's newer reasoning models (e.g. gpt-5.4 and later); for other providers it disables reasoning. `max` is the highest tier (above `xhigh`), supported by Anthropic models and OpenAI GPT-5.6 models. The gateway never downgrades effort tiers: providers that accept an effort parameter receive the value unchanged (an unsupported value results in a provider error), while providers that take a thinking budget instead (e.g. Anthropic, Google) translate each tier to a native budget. The exact values each provider mapping accepts are exposed as `reasoning_efforts` on `/v1/models`.", example: "medium", }), reasoning: z @@ -317,7 +317,7 @@ export const completionsRequestSchema = z.object({ .optional() .openapi({ description: - "Controls the reasoning effort. Alternative to top-level reasoning_effort. Cannot be used together with reasoning_effort. `max` is the highest tier (above `xhigh`), supported by Anthropic models and OpenAI GPT-5.6 models. Values are forwarded to the provider as-is; unsupported values result in a provider error. See `reasoning_efforts` on `/v1/models` for the values each mapping accepts.", + "Controls the reasoning effort. Alternative to top-level reasoning_effort. Cannot be used together with reasoning_effort. `max` is the highest tier (above `xhigh`), supported by Anthropic models and OpenAI GPT-5.6 models. Tiers are never downgraded by the gateway: enum-based providers receive the value unchanged (unsupported values result in a provider error), while budget-based providers (e.g. Anthropic, Google) translate the tier to a native thinking budget. See `reasoning_efforts` on `/v1/models` for the values each mapping accepts.", example: "medium", }), max_tokens: z.number().int().positive().optional().openapi({ diff --git a/apps/playground/src/components/playground/chat-page-client.tsx b/apps/playground/src/components/playground/chat-page-client.tsx index 02a3e3cb4b..d08da84807 100644 --- a/apps/playground/src/components/playground/chat-page-client.tsx +++ b/apps/playground/src/components/playground/chat-page-client.tsx @@ -51,7 +51,10 @@ import { getModelPreferenceCookie, setModelPreferenceCookie, } from "@/lib/model-preferences"; -import { getReasoningEffortOptions } from "@/lib/model-utils"; +import { + getFallbackReasoningEffortOptions, + getReasoningEffortOptions, +} from "@/lib/model-utils"; import { shouldDisableFallback } from "@/lib/no-fallback"; import { getErrorMessage } from "@/lib/utils"; @@ -1889,20 +1892,19 @@ export default function ChatPageClient({ setText(value); }; - // Reset reasoning effort when switching to a non-reasoning model, or when - // the new model declares its supported efforts and the current value isn't - // among them. + // Reset reasoning effort when switching to a model that doesn't support + // it. The effective options mirror what the selector renders: the model's + // declared efforts, or the generic fallback set when undeclared. useEffect(() => { if (!reasoningEffort) { return; } - if ( - !supportsReasoning || - (reasoningEfforts && !reasoningEfforts.includes(reasoningEffort)) - ) { + const effortOptions = + reasoningEfforts ?? getFallbackReasoningEffortOptions(selectedModel); + if (!supportsReasoning || !effortOptions.includes(reasoningEffort)) { setReasoningEffort(""); } - }, [supportsReasoning, reasoningEffort, reasoningEfforts]); + }, [supportsReasoning, reasoningEffort, reasoningEfforts, selectedModel]); // Reset image size/quality only when the selected model changes and the // current value is not valid for the new model. Including alibabaImageSize @@ -2691,20 +2693,19 @@ function ExtraChatPanel({ return getReasoningEffortOptions(mapping ? [mapping] : []); }, [models, selectedModel]); - // Reset reasoning effort when switching to a non-reasoning model, or when - // the new model declares its supported efforts and the current value isn't - // among them. + // Reset reasoning effort when switching to a model that doesn't support + // it. The effective options mirror what the selector renders: the model's + // declared efforts, or the generic fallback set when undeclared. useEffect(() => { if (!reasoningEffort) { return; } - if ( - !supportsReasoning || - (reasoningEfforts && !reasoningEfforts.includes(reasoningEffort)) - ) { + const effortOptions = + reasoningEfforts ?? getFallbackReasoningEffortOptions(selectedModel); + if (!supportsReasoning || !effortOptions.includes(reasoningEffort)) { setReasoningEffort(""); } - }, [supportsReasoning, reasoningEffort, reasoningEfforts]); + }, [supportsReasoning, reasoningEffort, reasoningEfforts, selectedModel]); const supportsWebSearch = useMemo(() => { if (!selectedModel) { diff --git a/apps/playground/src/components/playground/chat-ui.tsx b/apps/playground/src/components/playground/chat-ui.tsx index d08e62a0cc..7f6556d2da 100644 --- a/apps/playground/src/components/playground/chat-ui.tsx +++ b/apps/playground/src/components/playground/chat-ui.tsx @@ -101,6 +101,7 @@ import { parsePlaygroundMessageMetadata, type PlaygroundMessageMetadata, } from "@/lib/message-metadata"; +import { getFallbackReasoningEffortOptions } from "@/lib/model-utils"; import { cn } from "@/lib/utils"; import type { ReasoningEffortOption } from "@/lib/fetch-models"; @@ -1888,11 +1889,7 @@ export const ChatUI = ({ Auto {( reasoningEfforts ?? - // Fallback for models that don't declare their - // supported efforts in the catalog yet. - ((selectedModel.includes("gpt-5") - ? ["minimal", "low", "medium", "high"] - : ["low", "medium", "high"]) as ReasoningEffortOption[]) + getFallbackReasoningEffortOptions(selectedModel) ).map((effort) => ( {REASONING_EFFORT_LABELS[effort]} diff --git a/apps/playground/src/lib/model-utils.ts b/apps/playground/src/lib/model-utils.ts index 0f333ae258..ff5b245db1 100644 --- a/apps/playground/src/lib/model-utils.ts +++ b/apps/playground/src/lib/model-utils.ts @@ -15,6 +15,19 @@ export const REASONING_EFFORT_ORDER: ReasoningEffortOption[] = [ "max", ]; +/** + * Generic effort options for models that don't declare their supported + * values in the catalog yet. Used for both rendering the selector and + * resetting a stale selection, so the two never disagree. + */ +export function getFallbackReasoningEffortOptions( + selectedModel: string, +): ReasoningEffortOption[] { + return selectedModel.includes("gpt-5") + ? ["minimal", "low", "medium", "high"] + : ["low", "medium", "high"]; +} + /** * Union of the reasoning_effort values declared by the given provider * mappings, in ascending order of effort. Returns null when none of the