From e8ab7682a554530e128cc45da024e86b5d4941e7 Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 12:48:13 -0700 Subject: [PATCH 1/9] Add model-level reasoning effort metadata Introduce per-model reasoning control metadata on catalog entries and model descriptors so /effort support can be expanded without provider-wide inference. Resolve /effort through explicit model metadata first, preserve legacy allowlist behavior, and treat supportsReasoning-only entries as capability metadata that does not mutate requests. Guard OpenAI shim effort serialization with the model-level wire support check and add focused tests for capability-only, explicit metadata, opt-out, and toggle-mode cases. Document the reasoning metadata contract and provider follow-up workflow. --- docs/integrations/reasoning-effort.md | 56 ++++++ src/integrations/descriptors.ts | 21 +++ src/services/api/client.ts | 4 +- src/utils/effort.codex.test.ts | 180 +++++++++++++++++++ src/utils/effort.ts | 238 ++++++++++++++++++++++++-- 5 files changed, 487 insertions(+), 12 deletions(-) create mode 100644 docs/integrations/reasoning-effort.md diff --git a/docs/integrations/reasoning-effort.md b/docs/integrations/reasoning-effort.md new file mode 100644 index 0000000000..6aafe8bbad --- /dev/null +++ b/docs/integrations/reasoning-effort.md @@ -0,0 +1,56 @@ +# Reasoning and /effort Metadata + +OpenClaude treats reasoning support as a per-model capability. Provider and gateway catalogs can contain a mix of reasoning and non-reasoning models, so reasoning controls must never be inferred provider-wide. + +## Concepts + +`capabilities.supportsReasoning` means the model is known to support reasoning or thinking behavior. It is safe capability metadata, but by itself it does not authorize OpenClaude to mutate API requests. + +`reasoning` describes the request control surface OpenClaude can safely use for that exact model entry or model descriptor. + +```ts +reasoning: { + mode: 'levels' | 'toggle' | 'always-on' + levels?: ['low', 'medium', 'high', 'xhigh', 'max'] + defaultLevel?: 'low' | 'medium' | 'high' | 'xhigh' | 'max' + wireFormat?: + | 'reasoning_effort' + | 'reasoning_object' + | 'thinking_type' + | 'deepseek_compatible' + | 'zai_compatible' + | 'none' + disableFormat?: 'thinking_type_disabled' +} +``` + +## Backward Compatibility + +The `/effort` resolver is intentionally conservative: + +1. Explicit per-model `reasoning` metadata wins. +2. Existing hardcoded legacy effort support remains unchanged. +3. `supportsReasoning: true` without `reasoning` metadata is treated as reasoning-capable but not controllable. +4. Unknown models do not receive new reasoning request fields. + +This means existing OpenAI, Codex, Claude, Gemini, and configured 3P override behavior remains active, while catalogs can safely mark models with `supportsReasoning` before their exact request shape has been audited. + +## Provider and Gateway Rules + +Annotate reasoning per exact model on the route where it was verified. Aggregating gateways must not add reasoning controls at the provider level because different upstream models accept different parameters and levels. + +Prefer catalog-entry metadata when a gateway route differs from the canonical model descriptor. For example, a model may support reasoning directly from its vendor but reject `reasoning_effort` through a gateway. + +Use `mode: 'always-on'` with `wireFormat: 'none'` for models that emit reasoning but do not have a verified control parameter on that route. + +## Adding Support + +Before adding `reasoning` metadata for a model: + +1. Probe the exact route and model ID OpenClaude will send. +2. Record accepted levels and rejected levels. +3. Check whether disabling thinking is supported and what request shape is required. +4. Confirm whether accepted parameters actually change behavior or are silent no-ops. +5. Add focused tests for the resolver and request serialization path. + +Do not use `supportsReasoning: true` alone as evidence that `reasoning_effort` or any other effort field is accepted. \ No newline at end of file diff --git a/src/integrations/descriptors.ts b/src/integrations/descriptors.ts index 8873d8f4fb..91fe43be18 100644 --- a/src/integrations/descriptors.ts +++ b/src/integrations/descriptors.ts @@ -55,6 +55,25 @@ export interface CapabilityFlags { supportsEmbeddings?: boolean } +export type ReasoningControlMode = 'levels' | 'toggle' | 'always-on' +export type ReasoningEffortLevel = 'low' | 'medium' | 'high' | 'xhigh' | 'max' +export type ReasoningWireFormat = + | 'reasoning_effort' + | 'reasoning_object' + | 'thinking_type' + | 'deepseek_compatible' + | 'zai_compatible' + | 'none' +export type ReasoningDisableFormat = 'thinking_type_disabled' + +export interface ReasoningControlMetadata { + mode: ReasoningControlMode + levels?: ReasoningEffortLevel[] + defaultLevel?: ReasoningEffortLevel + wireFormat?: ReasoningWireFormat + disableFormat?: ReasoningDisableFormat +} + export interface TransportConfig { kind: TransportKind headers?: Record @@ -84,6 +103,7 @@ export interface ModelCatalogEntry { hidden?: boolean modelDescriptorId?: string capabilities?: CapabilityFlags + reasoning?: ReasoningControlMetadata contextWindow?: number maxOutputTokens?: number transportOverrides?: CatalogTransportOverrides @@ -311,6 +331,7 @@ export interface ModelDescriptor { defaultModel: string providerModelMap?: Partial> capabilities: CapabilityFlags + reasoning?: ReasoningControlMetadata contextWindow?: number maxOutputTokens?: number cacheConfig?: CacheConfig diff --git a/src/services/api/client.ts b/src/services/api/client.ts index 3952fd968f..af2086f4ce 100644 --- a/src/services/api/client.ts +++ b/src/services/api/client.ts @@ -12,6 +12,7 @@ import { import { convertEffortValueToLevel, type EffortValue, + modelSupportsWireEffort, standardEffortToOpenAI, type OpenAIEffortLevel, } from 'src/utils/effort.js' @@ -329,8 +330,9 @@ export async function getAnthropicClient({ }): Promise { // Convert the runtime effort value to the OpenAI-shaped enum the shim // expects. Undefined → shim falls back to descriptor/alias defaults. + const effortModel = providerOverride?.model ?? model const shimReasoningEffort: OpenAIEffortLevel | undefined = - effortValue !== undefined + effortValue !== undefined && effortModel && modelSupportsWireEffort(effortModel) ? standardEffortToOpenAI(convertEffortValueToLevel(effortValue)) : undefined const containerId = process.env.CLAUDE_CODE_CONTAINER_ID diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index a1e4fd8dc9..b07cf8bf43 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -15,6 +15,7 @@ import * as actualThinking from './thinking.js' import * as actualGrowthbook from 'src/services/analytics/growthbook.js' import * as actualProviders from './model/providers.js' import * as actualModelSupportOverrides from './model/modelSupportOverrides.js' +import * as actualIntegrations from '../integrations/index.js' beforeEach(async () => { await acquireSharedMutationLock('utils/effort.codex.test.ts') @@ -31,6 +32,9 @@ afterEach(() => { async function importFreshEffortModule(options: { provider: 'codex' | 'openai' supportsCodexReasoningEffort: boolean + routeId?: string + catalogEntries?: any[] + modelDescriptors?: Record }) { mock.module('./model/providers.js', () => ({ ...actualProviders, @@ -44,6 +48,12 @@ async function importFreshEffortModule(options: { ...actualProviderConfig, supportsCodexReasoningEffort: () => options.supportsCodexReasoningEffort, })) + mock.module('../integrations/index.js', () => ({ + ...actualIntegrations, + resolveActiveRouteIdFromEnv: () => options.routeId, + getCatalogEntriesForRoute: () => options.catalogEntries ?? [], + getModel: (id: string) => options.modelDescriptors?.[id], + })) mock.module('./auth.js', () => ({ ...actualAuth, isProSubscriber: () => false, @@ -266,3 +276,173 @@ test('modelUsesOpenAIEffort: Claude/Gemini are excluded even on the openai provi // Standard branch: no OPENAI_EFFORT_LEVELS, just the supported standard levels expect(opusLevels).toEqual(['low', 'medium', 'high', 'xhigh', 'max']) }) + +test('supportsReasoning-only catalog entries do not enable effort or wire mutation', async () => { + const { + getAvailableEffortLevels, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'atlas-cloud', + catalogEntries: [ + { + id: 'moonshotai/kimi-k2.5', + apiName: 'moonshotai/kimi-k2.5', + capabilities: { supportsReasoning: true }, + }, + ], + }) + + expect(resolveModelReasoningControl('moonshotai/kimi-k2.5')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'capability', + }) + expect(modelSupportsEffort('moonshotai/kimi-k2.5')).toBe(false) + expect(modelSupportsWireEffort('moonshotai/kimi-k2.5')).toBe(false) + expect(getAvailableEffortLevels('moonshotai/kimi-k2.5')).toEqual([]) + expect(resolveAppliedEffort('moonshotai/kimi-k2.5', 'high')).toBeUndefined() +}) + +test('explicit reasoning metadata enables model-level effort without provider-wide inference', async () => { + const { + getAvailableEffortLevels, + getDefaultEffortForModel, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'atlas-cloud', + catalogEntries: [ + { + id: 'moonshotai/kimi-k2.6', + apiName: 'moonshotai/kimi-k2.6', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['low', 'medium', 'high'], + defaultLevel: 'medium', + wireFormat: 'reasoning_effort', + }, + }, + { + id: 'xai/grok-build-0.1', + apiName: 'xai/grok-build-0.1', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'always-on', + wireFormat: 'none', + }, + }, + ], + }) + + expect(resolveModelReasoningControl('moonshotai/kimi-k2.6')).toMatchObject({ + supportsReasoning: true, + controllable: true, + source: 'metadata', + levels: ['low', 'medium', 'high'], + defaultLevel: 'medium', + wireFormat: 'reasoning_effort', + }) + expect(modelSupportsEffort('moonshotai/kimi-k2.6')).toBe(true) + expect(modelSupportsWireEffort('moonshotai/kimi-k2.6')).toBe(true) + expect(getAvailableEffortLevels('moonshotai/kimi-k2.6')).toEqual([ + 'low', + 'medium', + 'high', + ]) + expect(getDefaultEffortForModel('moonshotai/kimi-k2.6')).toBe('medium') + expect(resolveAppliedEffort('moonshotai/kimi-k2.6', undefined)).toBe('medium') + expect(resolveAppliedEffort('moonshotai/kimi-k2.6', 'xhigh')).toBe('high') + + expect(resolveModelReasoningControl('xai/grok-build-0.1')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'metadata', + wireFormat: 'none', + }) + expect(modelSupportsEffort('xai/grok-build-0.1')).toBe(false) + expect(modelSupportsWireEffort('xai/grok-build-0.1')).toBe(false) + expect(resolveAppliedEffort('xai/grok-build-0.1', 'high')).toBeUndefined() +}) + +test('explicit non-controllable metadata opts out even when the model matches legacy rules', async () => { + const { + getAvailableEffortLevels, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: true, + routeId: 'custom-gateway', + catalogEntries: [ + { + id: 'gpt-5.4', + apiName: 'gpt-5.4', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'always-on', + wireFormat: 'none', + }, + }, + ], + }) + + expect(resolveModelReasoningControl('gpt-5.4')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'metadata', + wireFormat: 'none', + }) + expect(modelSupportsEffort('gpt-5.4')).toBe(false) + expect(modelSupportsWireEffort('gpt-5.4')).toBe(false) + expect(getAvailableEffortLevels('gpt-5.4')).toEqual([]) + expect(resolveAppliedEffort('gpt-5.4', 'high')).toBeUndefined() +}) + +test('toggle reasoning metadata stays non-controllable until toggle serialization exists', async () => { + const { + getAvailableEffortLevels, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'custom-gateway', + catalogEntries: [ + { + id: 'toggle-model', + apiName: 'toggle-model', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'toggle', + wireFormat: 'reasoning_effort', + }, + }, + ], + }) + + expect(resolveModelReasoningControl('toggle-model')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'metadata', + mode: 'toggle', + wireFormat: 'reasoning_effort', + }) + expect(modelSupportsEffort('toggle-model')).toBe(false) + expect(modelSupportsWireEffort('toggle-model')).toBe(false) + expect(getAvailableEffortLevels('toggle-model')).toEqual([]) + expect(resolveAppliedEffort('toggle-model', 'high')).toBeUndefined() +}) diff --git a/src/utils/effort.ts b/src/utils/effort.ts index 9af48a3bbe..9dbe35af7b 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -7,6 +7,16 @@ import { getAPIProvider } from './model/providers.js' import { get3PModelCapabilityOverride } from './model/modelSupportOverrides.js' import { getAntModelOverrideConfig, resolveAntModel } from './model/antModels.js' import { supportsCodexReasoningEffort } from '../services/api/providerConfig.js' +import { + getCatalogEntriesForRoute, + getModel, + resolveActiveRouteIdFromEnv, +} from '../integrations/index.js' +import type { + CapabilityFlags, + ReasoningControlMetadata, + ReasoningWireFormat, +} from '../integrations/descriptors.js' import { isEnvTruthy } from './envUtils.js' import type { EffortLevel } from 'src/entrypoints/sdk/runtimeTypes.js' @@ -30,8 +40,123 @@ export const OPENAI_EFFORT_LEVELS = [ export type OpenAIEffortLevel = typeof OPENAI_EFFORT_LEVELS[number] export type EffortValue = EffortLevel | number -// @[MODEL LAUNCH]: Add the new model to the allowlist if it supports the effort parameter. -export function modelSupportsEffort(model: string): boolean { +export type ReasoningControlResolution = { + supportsReasoning: boolean + controllable: boolean + mode?: ReasoningControlMetadata['mode'] + levels: EffortLevel[] + defaultLevel?: EffortValue + wireFormat?: ReasoningWireFormat + source: 'metadata' | 'capability' | 'legacy' | 'none' +} + +const DEFAULT_REASONING_LEVELS: EffortLevel[] = ['low', 'medium', 'high'] + +function isSupportedEffortLevel(level: string): level is EffortLevel { + return (EFFORT_LEVELS as readonly string[]).includes(level) +} + +function normalizeReasoningLevels( + levels: ReasoningControlMetadata['levels'] | undefined, +): EffortLevel[] { + const normalized = (levels ?? DEFAULT_REASONING_LEVELS).filter( + isSupportedEffortLevel, + ) + return normalized.length > 0 ? normalized : [...DEFAULT_REASONING_LEVELS] +} + +function normalizeReasoningDefaultLevel( + level: ReasoningControlMetadata['defaultLevel'] | undefined, + levels: EffortLevel[], +): EffortLevel | undefined { + if (!level || !isSupportedEffortLevel(level)) { + return undefined + } + return levels.includes(level) ? level : undefined +} + +function metadataWireFormatSupportsEffort( + wireFormat: ReasoningWireFormat | undefined, +): boolean { + // PR 1 only enables the existing OpenAI-compatible effort serializer. + // Other wire formats are typed for follow-up provider PRs, but must not + // silently flow through as top-level reasoning_effort before their request + // shapers are implemented. + return wireFormat === 'reasoning_effort' +} + +function resolveCatalogReasoningMetadata(model: string): { + capabilities?: CapabilityFlags + reasoning?: ReasoningControlMetadata +} | undefined { + const routeId = resolveActiveRouteIdFromEnv(process.env) + if (!routeId || routeId === 'anthropic') { + return undefined + } + + const normalizedModel = model.trim().split('?', 1)[0]!.trim().toLowerCase() + const entry = getCatalogEntriesForRoute(routeId).find(catalogEntry => + catalogEntry.apiName.trim().toLowerCase() === normalizedModel || + catalogEntry.id.trim().toLowerCase() === normalizedModel, + ) + + if (!entry) { + return undefined + } + + const descriptor = entry.modelDescriptorId + ? getModel(entry.modelDescriptorId) + : undefined + + return { + capabilities: entry.capabilities ?? descriptor?.capabilities, + reasoning: entry.reasoning ?? descriptor?.reasoning, + } +} + +function resolveMetadataReasoningControl( + model: string, +): ReasoningControlResolution | undefined { + const metadata = resolveCatalogReasoningMetadata(model) + if (!metadata) { + return undefined + } + + const { capabilities, reasoning } = metadata + if (!reasoning) { + return capabilities?.supportsReasoning === undefined + ? undefined + : { + supportsReasoning: capabilities.supportsReasoning, + controllable: false, + levels: [], + source: 'capability', + } + } + + const levels = reasoning.mode === 'levels' + ? normalizeReasoningLevels(reasoning.levels) + : [] + const wireFormat = reasoning.wireFormat + const controllable = Boolean( + capabilities?.supportsReasoning !== false && + metadataWireFormatSupportsEffort(wireFormat) && + reasoning.mode === 'levels' && + levels.length > 0, + ) + + return { + supportsReasoning: capabilities?.supportsReasoning ?? true, + controllable, + mode: reasoning.mode, + levels, + defaultLevel: normalizeReasoningDefaultLevel(reasoning.defaultLevel, levels), + wireFormat, + source: 'metadata', + } +} + +function legacyModelSupportsEffort(model: string): boolean { const m = model.toLowerCase() if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { return true @@ -74,9 +199,64 @@ export function modelSupportsEffort(model: string): boolean { return getAPIProvider() === 'firstParty' } +function resolveLegacyReasoningControl(model: string): ReasoningControlResolution { + if (!legacyModelSupportsEffort(model)) { + return { + supportsReasoning: false, + controllable: false, + levels: [], + source: 'none', + } + } + + return { + supportsReasoning: true, + controllable: true, + mode: 'levels', + levels: getLegacyAvailableEffortLevels(model), + defaultLevel: getLegacyDefaultEffortForModel(model), + wireFormat: 'reasoning_effort', + source: 'legacy', + } +} + +export function resolveModelReasoningControl( + model: string, +): ReasoningControlResolution { + const metadata = resolveMetadataReasoningControl(model) + if (metadata) { + return metadata + } + + return resolveLegacyReasoningControl(model) +} + +// @[MODEL LAUNCH]: Add the new model to the allowlist if it supports the effort parameter. +export function modelSupportsEffort(model: string): boolean { + if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { + return true + } + const supported3P = get3PModelCapabilityOverride(model, 'effort') + if (supported3P !== undefined) { + return supported3P + } + return resolveModelReasoningControl(model).controllable +} + +export function modelSupportsWireEffort(model: string): boolean { + if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { + return true + } + const supported3P = get3PModelCapabilityOverride(model, 'effort') + if (supported3P !== undefined) { + return supported3P + } + const control = resolveModelReasoningControl(model) + return Boolean(control.controllable && metadataWireFormatSupportsEffort(control.wireFormat)) +} // @[MODEL LAUNCH]: Add the new model to the allowlist if it supports 'max' effort. // Per API docs, 'max' is Opus 4.6 only for public models — other models return an error. -export function modelSupportsMaxEffort(model: string): boolean { +function legacyModelSupportsMaxEffort(model: string): boolean { const supported3P = get3PModelCapabilityOverride(model, 'max_effort') if (supported3P !== undefined) { return supported3P @@ -93,8 +273,8 @@ export function modelSupportsMaxEffort(model: string): boolean { // @[MODEL LAUNCH]: Add the new model to the allowlist if it supports 'xhigh' effort. // xhigh is reserved for OpenAI/Codex models and OpenCode Claude opus 4-7 / 4-8. // All other effort-supporting models reject xhigh at the API. -export function modelSupportsXHighEffort(model: string): boolean { - if (!modelSupportsEffort(model)) { +function legacyModelSupportsXHighEffort(model: string): boolean { + if (!legacyModelSupportsEffort(model)) { return false } const supported3P = get3PModelCapabilityOverride(model, 'xhigh_effort') @@ -133,8 +313,8 @@ export function modelUsesOpenAIEffort(model: string): boolean { return true } -export function getAvailableEffortLevels(model: string): EffortLevel[] { - if (!modelSupportsEffort(model)) { +function getLegacyAvailableEffortLevels(model: string): EffortLevel[] { + if (!legacyModelSupportsEffort(model)) { return [] } // OpenCode Claude and Gemini models use /messages or /models/gemini-* @@ -150,15 +330,38 @@ export function getAvailableEffortLevels(model: string): EffortLevel[] { return [...OPENAI_EFFORT_LEVELS] as EffortLevel[] } const levels: EffortLevel[] = ['low', 'medium', 'high'] - if (modelSupportsXHighEffort(model)) { + if (legacyModelSupportsXHighEffort(model)) { levels.push('xhigh') } - if (modelSupportsMaxEffort(model)) { + if (legacyModelSupportsMaxEffort(model)) { levels.push('max') } return levels } +export function modelSupportsMaxEffort(model: string): boolean { + const control = resolveModelReasoningControl(model) + if (control.source === 'metadata' || control.source === 'capability') { + return control.levels.includes('max') + } + return legacyModelSupportsMaxEffort(model) +} + +export function modelSupportsXHighEffort(model: string): boolean { + const control = resolveModelReasoningControl(model) + if (control.source === 'metadata' || control.source === 'capability') { + return control.levels.includes('xhigh') + } + return legacyModelSupportsXHighEffort(model) +} + +export function getAvailableEffortLevels(model: string): EffortLevel[] { + const control = resolveModelReasoningControl(model) + if (control.source === 'metadata' || control.source === 'capability') { + return [...control.levels] + } + return getLegacyAvailableEffortLevels(model) +} export function getEffortLevelLabel(level: EffortLevel | OpenAIEffortLevel): string { if (level === 'xhigh') return 'Extra High' if (level === 'max') return 'Max' @@ -272,6 +475,10 @@ export function resolveAppliedEffort( if (envOverride === null) { return undefined } + if (!modelSupportsEffort(model)) { + return undefined + } + const resolved = envOverride ?? appStateEffortValue ?? getDefaultEffortForModel(model) // API rejects 'max' on non-Opus-4.6 Anthropic models — downgrade to 'high'. @@ -342,6 +549,15 @@ export function convertEffortValueToLevel(value: EffortValue): EffortLevel { return 'high' } +export function getDefaultEffortForModel( + model: string, +): EffortValue | undefined { + const control = resolveModelReasoningControl(model) + if (control.source === 'metadata' || control.source === 'capability') { + return control.defaultLevel + } + return getLegacyDefaultEffortForModel(model) +} /** * Get user-facing description for effort levels * @@ -405,7 +621,7 @@ export function getOpusDefaultEffortConfig(): OpusDefaultEffortConfig { } // @[MODEL LAUNCH]: Update the default effort levels for new models -export function getDefaultEffortForModel( +function getLegacyDefaultEffortForModel( model: string, ): EffortValue | undefined { if (process.env.USER_TYPE === 'ant') { @@ -448,7 +664,7 @@ export function getDefaultEffortForModel( } // When ultrathink feature is on, default effort to medium (ultrathink bumps to high) - if (isUltrathinkEnabled() && modelSupportsEffort(model)) { + if (isUltrathinkEnabled() && legacyModelSupportsEffort(model)) { return 'medium' } From 94cfa5e024d7ac65e7471ea8f33af43d0300c694 Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 14:31:51 -0700 Subject: [PATCH 2/9] Expand reasoning effort routing Centralize OpenAI shim reasoning request planning so DeepSeek-compatible and Z.AI-compatible controls flow through the effort resolver instead of provider-specific shim helpers. Add compatibility metadata handling for DeepSeek and Z.AI routes while keeping supportsReasoning-only catalog entries non-controllable until exact wire formats are verified. Respect route removeBodyFields after compatibility serialization and make provider override support checks use the resolved override route/base URL instead of ambient provider metadata. Document temporary compatibility rules and add regression coverage for Atlas DeepSeek, Z.AI levels, Groq stripping, providerOverride OpenAI effort, providerOverride Groq stripping, and non-generic metadata opt-out. Verified: bun test --feature=UNATTENDED_RETRY src/utils/effort.codex.test.ts; bun test --feature=UNATTENDED_RETRY src/services/api/client.test.ts; bun test --feature=UNATTENDED_RETRY src/services/api/openaiShim.test.ts; node .\\node_modules\\typescript\\bin\\tsc --noEmit; bun run build; git diff --check. --- docs/integrations/reasoning-effort.md | 2 + src/integrations/runtimeMetadata.ts | 4 +- src/services/api/client.test.ts | 109 +++++++++ src/services/api/client.ts | 26 +- src/services/api/openaiShim.test.ts | 35 +++ src/services/api/openaiShim.ts | 98 +++----- src/utils/effort.codex.test.ts | 212 +++++++++++++++- src/utils/effort.ts | 340 ++++++++++++++++++++++++-- 8 files changed, 742 insertions(+), 84 deletions(-) diff --git a/docs/integrations/reasoning-effort.md b/docs/integrations/reasoning-effort.md index 6aafe8bbad..d307f5a94c 100644 --- a/docs/integrations/reasoning-effort.md +++ b/docs/integrations/reasoning-effort.md @@ -35,6 +35,8 @@ The `/effort` resolver is intentionally conservative: This means existing OpenAI, Codex, Claude, Gemini, and configured 3P override behavior remains active, while catalogs can safely mark models with `supportsReasoning` before their exact request shape has been audited. +A temporary compatibility layer also preserves verified request shaping that existed before per-model `reasoning` metadata. For example, DeepSeek-compatible routes can still map `/effort xhigh` to provider `reasoning_effort: "max"`, and Z.AI GLM routes can still map supported controls through their `thinking` request shape. These compatibility rules are intentionally centralized in the effort resolver so they can be removed as catalogs gain explicit `reasoning` metadata. + ## Provider and Gateway Rules Annotate reasoning per exact model on the route where it was verified. Aggregating gateways must not add reasoning controls at the provider level because different upstream models accept different parameters and levels. diff --git a/src/integrations/runtimeMetadata.ts b/src/integrations/runtimeMetadata.ts index bc8cf8a707..8d5e8c6b2b 100644 --- a/src/integrations/runtimeMetadata.ts +++ b/src/integrations/runtimeMetadata.ts @@ -220,6 +220,7 @@ export function resolveOpenAIShimRuntimeContext(options?: { model?: string activeProfileProvider?: string treatAsLocal?: boolean + preferBaseUrlRoute?: boolean }): OpenAIShimRuntimeContext { const processEnv = options?.processEnv ?? process.env const runtimeEnv: NodeJS.ProcessEnv = { @@ -240,7 +241,8 @@ export function resolveOpenAIShimRuntimeContext(options?: { const baseUrlRouteId = resolveRouteIdFromBaseUrl(options?.baseUrl) const routeId = baseUrlRouteId && - (!activeRouteId || activeRouteId === 'anthropic' || activeRouteId === 'openai') + (options?.preferBaseUrlRoute || + !activeRouteId || activeRouteId === 'anthropic' || activeRouteId === 'openai') ? baseUrlRouteId : activeRouteId const descriptor = diff --git a/src/services/api/client.test.ts b/src/services/api/client.test.ts index 6da43c012b..e6b9d37178 100644 --- a/src/services/api/client.test.ts +++ b/src/services/api/client.test.ts @@ -1332,6 +1332,115 @@ test('strips Anthropic-specific custom headers on providerOverride shim requests expect(capturedHeaders?.get('authorization')).toBe('Bearer provider-test-key') }) +test('providerOverride OpenAI gpt effort does not fall back to ambient provider', async () => { + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + + return new Response( + JSON.stringify({ + id: 'chatcmpl-provider-override-openai', + model: 'gpt-5.4', + choices: [ + { + message: { + role: 'assistant', + content: 'ok', + }, + finish_reason: 'stop', + }, + ], + usage: { + prompt_tokens: 8, + completion_tokens: 3, + total_tokens: 11, + }, + }), + { + headers: { + 'Content-Type': 'application/json', + }, + }, + ) + }) as FetchType + + const client = (await getAnthropicClient({ + maxRetries: 0, + effortValue: 'xhigh', + providerOverride: { + model: 'gpt-5.4', + baseURL: 'https://api.openai.com/v1', + apiKey: 'provider-test-key', + }, + })) as unknown as ShimClient + + await client.beta.messages.create({ + model: 'unused', + system: 'test system', + messages: [{ role: 'user', content: 'hello' }], + max_tokens: 64, + stream: false, + }) + + expect(requestBody?.reasoning_effort).toBe('xhigh') +}) +test('providerOverride Groq DeepSeek does not receive stripped effort override', async () => { + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + + return new Response( + JSON.stringify({ + id: 'chatcmpl-provider-override-groq', + model: 'deepseek-r1-distill-llama-70b', + choices: [ + { + message: { + role: 'assistant', + content: 'ok', + }, + finish_reason: 'stop', + }, + ], + usage: { + prompt_tokens: 8, + completion_tokens: 3, + total_tokens: 11, + }, + }), + { + headers: { + 'Content-Type': 'application/json', + }, + }, + ) + }) as FetchType + + const client = (await getAnthropicClient({ + maxRetries: 0, + effortValue: 'xhigh', + providerOverride: { + model: 'deepseek-r1-distill-llama-70b', + baseURL: 'https://api.groq.com/openai/v1', + apiKey: 'provider-test-key', + }, + })) as unknown as ShimClient + + await client.beta.messages.create({ + model: 'unused', + system: 'test system', + messages: [{ role: 'user', content: 'hello' }], + max_tokens: 64, + stream: false, + thinking: { type: 'enabled' }, + }) + + expect(requestBody?.thinking).toEqual({ type: 'enabled' }) + expect(requestBody?.reasoning_effort).toBeUndefined() + expect(requestBody?.store).toBeUndefined() +}) test('rejects CRLF-injected custom headers before sending OpenAI-compatible shim requests', async () => { let capturedHeaders: Headers | undefined diff --git a/src/services/api/client.ts b/src/services/api/client.ts index af2086f4ce..6a49104840 100644 --- a/src/services/api/client.ts +++ b/src/services/api/client.ts @@ -12,6 +12,7 @@ import { import { convertEffortValueToLevel, type EffortValue, + modelSupportsShimReasoningEffort, modelSupportsWireEffort, standardEffortToOpenAI, type OpenAIEffortLevel, @@ -45,6 +46,7 @@ import { getXiaomiMimoBaseUrlOverride, resolveEnvOnlyProviderRouteId, } from '../../integrations/routeMetadata.js' +import { resolveOpenAIShimRuntimeContext } from '../../integrations/runtimeMetadata.js' import { shouldUseFirstPartyAnthropicAuth, type ProviderOverride, @@ -331,8 +333,30 @@ export async function getAnthropicClient({ // Convert the runtime effort value to the OpenAI-shaped enum the shim // expects. Undefined → shim falls back to descriptor/alias defaults. const effortModel = providerOverride?.model ?? model + const providerOverrideRuntimeContext = providerOverride && effortModel + ? resolveOpenAIShimRuntimeContext({ + processEnv: process.env, + baseUrl: providerOverride.baseURL, + model: effortModel, + preferBaseUrlRoute: true, + }) + : undefined + const providerOverrideShimConfig = providerOverrideRuntimeContext?.openaiShimConfig + const supportsShimReasoningEffort = effortModel + ? providerOverrideShimConfig + ? modelSupportsShimReasoningEffort( + effortModel, + providerOverrideShimConfig.thinkingRequestFormat, + providerOverrideShimConfig.removeBodyFields, + { + routeId: providerOverrideRuntimeContext?.routeId, + useRuntimeFallback: false, + }, + ) + : modelSupportsWireEffort(effortModel) + : false const shimReasoningEffort: OpenAIEffortLevel | undefined = - effortValue !== undefined && effortModel && modelSupportsWireEffort(effortModel) + effortValue !== undefined && supportsShimReasoningEffort ? standardEffortToOpenAI(convertEffortValueToLevel(effortValue)) : undefined const containerId = process.env.CLAUDE_CODE_CONTAINER_ID diff --git a/src/services/api/openaiShim.test.ts b/src/services/api/openaiShim.test.ts index 6a3ef5c06d..23787ba856 100644 --- a/src/services/api/openaiShim.test.ts +++ b/src/services/api/openaiShim.test.ts @@ -6114,6 +6114,41 @@ test('Groq: keeps max_completion_tokens and strips unsupported store', async () expect(requestBody?.store).toBeUndefined() }) + +test('Groq: strips reasoning_effort even when compat inference matches the model', async () => { + process.env.OPENAI_BASE_URL = 'https://api.groq.com/openai/v1' + process.env.OPENAI_API_KEY = 'gsk-test' + + let requestBody: Record | undefined + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + return new Response( + JSON.stringify({ + id: 'chatcmpl-1', + model: 'deepseek-r1-distill-llama-70b', + choices: [ + { message: { role: 'assistant', content: 'ok' }, finish_reason: 'stop' }, + ], + usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 }, + }), + { headers: { 'Content-Type': 'application/json' } }, + ) + }) as unknown as FetchType + + const client = createOpenAIShimClient({ reasoningEffort: 'xhigh' }) as OpenAIShimClient + await client.beta.messages.create({ + model: 'deepseek-r1-distill-llama-70b', + system: 'you are groq', + messages: [{ role: 'user', content: 'hi' }], + max_tokens: 256, + stream: false, + thinking: { type: 'enabled' }, + }) + + expect(requestBody?.thinking).toEqual({ type: 'enabled' }) + expect(requestBody?.reasoning_effort).toBeUndefined() + expect(requestBody?.store).toBeUndefined() +}) test('Moonshot: echoes reasoning_content on assistant tool-call messages', async () => { // Regression for: "API Error: 400 {"error":{"message":"thinking is enabled // but reasoning_content is missing in assistant tool call message at index diff --git a/src/services/api/openaiShim.ts b/src/services/api/openaiShim.ts index 7202097c69..69c9d14c74 100644 --- a/src/services/api/openaiShim.ts +++ b/src/services/api/openaiShim.ts @@ -38,6 +38,7 @@ import { } from '../../utils/codexCredentials.js' import { logForDebugging } from '../../utils/debug.js' import { isBareMode, isEnvTruthy } from '../../utils/envUtils.js' +import { resolveOpenAIShimReasoningRequestPlan } from '../../utils/effort.js' import { resolveGeminiCredential } from '../../utils/geminiAuth.js' import { hydrateGeminiAccessTokenFromSecureStorage } from '../../utils/geminiCredentials.js' import { hydrateGithubModelsTokenFromSecureStorage } from '../../utils/githubModelsCredentials.js' @@ -209,36 +210,6 @@ function hasCerebrasApiHost(baseUrl: string | undefined): boolean { } } -function normalizeDeepSeekReasoningEffort( - effort: 'low' | 'medium' | 'high' | 'xhigh', -): 'high' | 'max' { - return effort === 'xhigh' ? 'max' : 'high' -} - -function normalizeZaiReasoningEffort( - effort: 'low' | 'medium' | 'high' | 'xhigh', -): 'high' | 'max' { - return effort === 'xhigh' ? 'max' : 'high' -} - -function supportsZaiReasoningEffort(model: string | undefined): boolean { - const normalized = model?.trim().split('?', 1)[0]?.trim().toLowerCase() - return normalized === 'glm-5.2' -} - -function normalizeThinkingType( - value: string | undefined, -): 'enabled' | 'disabled' | undefined { - const normalized = value?.trim().toLowerCase() - if (normalized === 'disabled') { - return 'disabled' - } - if (normalized === 'enabled' || normalized === 'adaptive') { - return 'enabled' - } - return undefined -} - function formatRetryAfterHint(response: Response): string { const ra = response.headers.get('retry-after') return ra ? ` (Retry-After: ${ra})` : '' @@ -2439,6 +2410,7 @@ class OpenAIShimMessages { baseUrl: request.baseUrl, model: request.resolvedModel, treatAsLocal: isLocalProviderUrl(request.baseUrl), + preferBaseUrlRoute: Boolean(this.providerOverride), }) const shimConfig = runtimeShimContext.openaiShimConfig // When endpointPath is overridden, the body format must match the target @@ -2462,6 +2434,14 @@ class OpenAIShimMessages { ), }) + const reasoningRequestPlan = resolveOpenAIShimReasoningRequestPlan({ + model: request.resolvedModel, + requestedEffort: request.reasoning?.effort, + requestThinkingType: (params.thinking as { type?: string } | undefined)?.type, + defaultThinkingType: request.thinking?.type, + thinkingRequestFormat: shimConfig.thinkingRequestFormat, + }) + const body: Record = { model: request.resolvedModel, messages: openaiMessages, @@ -2472,8 +2452,8 @@ class OpenAIShimMessages { // request carries a reasoning effort (set via /effort, model alias default, // or `?reasoning=` query on the model string). OpenAI, Codex, and // most OpenAI-compatible endpoints read it from this top-level field. - if (request.reasoning) { - body.reasoning_effort = request.reasoning.effort + if (reasoningRequestPlan.wireFormat === 'reasoning_effort' && reasoningRequestPlan.reasoningEffort) { + body.reasoning_effort = reasoningRequestPlan.reasoningEffort } // Convert max_tokens to max_completion_tokens for OpenAI API compatibility. // Azure OpenAI requires max_completion_tokens and does not accept max_tokens. @@ -2527,48 +2507,32 @@ class OpenAIShimMessages { if (params.temperature !== undefined) body.temperature = params.temperature if (params.top_p !== undefined) body.top_p = params.top_p - if (shimConfig.thinkingRequestFormat === 'deepseek-compatible') { - const requestedThinkingType = (params.thinking as { type?: string } | undefined)?.type - const deepSeekThinkingType = - normalizeThinkingType(requestedThinkingType) - - if (deepSeekThinkingType) { - body.thinking = { type: deepSeekThinkingType } + if (reasoningRequestPlan.wireFormat === 'deepseek_compatible') { + if (reasoningRequestPlan.thinkingType) { + body.thinking = { type: reasoningRequestPlan.thinkingType } } - - if (deepSeekThinkingType === 'enabled') { - const effort = request.reasoning?.effort - if (effort) { - body.reasoning_effort = normalizeDeepSeekReasoningEffort(effort) - } + if (reasoningRequestPlan.reasoningEffort) { + body.reasoning_effort = reasoningRequestPlan.reasoningEffort } } - if (shimConfig.thinkingRequestFormat === 'zai-compatible') { - const requestedThinkingType = (params.thinking as { type?: string } | undefined)?.type - const zaiThinkingType = - normalizeThinkingType(requestedThinkingType) ?? - normalizeThinkingType(request.thinking?.type) - const zaiSupportsReasoningEffort = supportsZaiReasoningEffort( - request.resolvedModel, - ) - - if (zaiThinkingType === 'disabled') { - body.thinking = { type: 'disabled' } + if (reasoningRequestPlan.wireFormat === 'zai_compatible') { + if (reasoningRequestPlan.thinkingType) { + body.thinking = { type: reasoningRequestPlan.thinkingType } + } + if (reasoningRequestPlan.thinkingType === 'disabled') { + delete body.reasoning_effort + } else if (reasoningRequestPlan.reasoningEffort) { + body.reasoning_effort = reasoningRequestPlan.reasoningEffort + } else { delete body.reasoning_effort - } else if (zaiThinkingType === 'enabled' || request.reasoning?.effort) { - body.thinking = { type: 'enabled' } } + } - if (zaiThinkingType !== 'disabled' && request.reasoning?.effort) { - if (zaiSupportsReasoningEffort) { - body.reasoning_effort = normalizeZaiReasoningEffort( - request.reasoning.effort, - ) - } else { - delete body.reasoning_effort - } - } + // Route/model strip rules are authoritative even when compatibility + // serializers add provider-specific reasoning fields later in the pipeline. + for (const field of shimConfig.removeBodyFields ?? []) { + delete body[field] } if (params.tools && params.tools.length > 0) { diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index b07cf8bf43..5b494706b1 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -16,6 +16,7 @@ import * as actualGrowthbook from 'src/services/analytics/growthbook.js' import * as actualProviders from './model/providers.js' import * as actualModelSupportOverrides from './model/modelSupportOverrides.js' import * as actualIntegrations from '../integrations/index.js' +import * as actualRuntimeMetadata from '../integrations/runtimeMetadata.js' beforeEach(async () => { await acquireSharedMutationLock('utils/effort.codex.test.ts') @@ -35,6 +36,7 @@ async function importFreshEffortModule(options: { routeId?: string catalogEntries?: any[] modelDescriptors?: Record + openaiShimConfig?: any }) { mock.module('./model/providers.js', () => ({ ...actualProviders, @@ -51,9 +53,19 @@ async function importFreshEffortModule(options: { mock.module('../integrations/index.js', () => ({ ...actualIntegrations, resolveActiveRouteIdFromEnv: () => options.routeId, - getCatalogEntriesForRoute: () => options.catalogEntries ?? [], + getCatalogEntriesForRoute: (routeId: string) => + routeId === options.routeId ? (options.catalogEntries ?? []) : [], getModel: (id: string) => options.modelDescriptors?.[id], })) + mock.module('../integrations/runtimeMetadata.js', () => ({ + ...actualRuntimeMetadata, + resolveOpenAIShimRuntimeContext: () => ({ + routeId: options.routeId ?? null, + descriptor: null, + catalogEntry: null, + openaiShimConfig: options.openaiShimConfig ?? {}, + }), + })) mock.module('./auth.js', () => ({ ...actualAuth, isProSubscriber: () => false, @@ -446,3 +458,201 @@ test('toggle reasoning metadata stays non-controllable until toggle serializatio expect(getAvailableEffortLevels('toggle-model')).toEqual([]) expect(resolveAppliedEffort('toggle-model', 'high')).toBeUndefined() }) + +test('compat DeepSeek routes can use /effort without catalog reasoning metadata', async () => { + const { + getAvailableEffortLevels, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'atlas-cloud', + catalogEntries: [ + { + id: 'deepseek-ai/deepseek-v3.2', + apiName: 'deepseek-ai/deepseek-v3.2', + capabilities: { supportsReasoning: true }, + }, + ], + }) + + expect(resolveModelReasoningControl('deepseek-ai/deepseek-v3.2')).toMatchObject({ + supportsReasoning: true, + controllable: true, + source: 'compat', + wireFormat: 'deepseek_compatible', + }) + expect(modelSupportsEffort('deepseek-ai/deepseek-v3.2')).toBe(true) + expect(modelSupportsWireEffort('deepseek-ai/deepseek-v3.2')).toBe(true) + expect(getAvailableEffortLevels('deepseek-ai/deepseek-v3.2')).toEqual([ + 'low', + 'medium', + 'high', + 'xhigh', + ]) + expect(resolveAppliedEffort('deepseek-ai/deepseek-v3.2', 'xhigh')).toBe('xhigh') +}) + +test('compat DeepSeek routes stay non-controllable when the runtime shim strips reasoning_effort', async () => { + const { + getAvailableEffortLevels, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'groq', + openaiShimConfig: { + thinkingRequestFormat: 'deepseek-compatible', + removeBodyFields: ['store', 'reasoning_effort'], + }, + }) + + expect(resolveModelReasoningControl('deepseek-r1-distill-llama-70b')).toMatchObject({ + supportsReasoning: false, + controllable: false, + source: 'none', + }) + expect(modelSupportsEffort('deepseek-r1-distill-llama-70b')).toBe(false) + expect(modelSupportsWireEffort('deepseek-r1-distill-llama-70b')).toBe(false) + expect(getAvailableEffortLevels('deepseek-r1-distill-llama-70b')).toEqual([]) + expect(resolveAppliedEffort('deepseek-r1-distill-llama-70b', 'xhigh')).toBeUndefined() +}) + +test('compat Z.AI routes expose only verified levels and clamp stale values', async () => { + const { + getAvailableEffortLevels, + modelSupportsEffort, + modelSupportsWireEffort, + resolveAppliedEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'zai', + }) + + expect(resolveModelReasoningControl('glm-5.2')).toMatchObject({ + controllable: true, + source: 'compat', + wireFormat: 'zai_compatible', + levels: ['high', 'xhigh'], + }) + expect(getAvailableEffortLevels('glm-5.2')).toEqual(['high', 'xhigh']) + expect(resolveAppliedEffort('glm-5.2', 'low')).toBe('high') + expect(resolveAppliedEffort('glm-5.2', 'xhigh')).toBe('xhigh') + + expect(resolveModelReasoningControl('GLM-5.1')).toMatchObject({ + controllable: true, + source: 'compat', + wireFormat: 'zai_compatible', + levels: ['high'], + }) + expect(modelSupportsEffort('GLM-5.1')).toBe(true) + expect(modelSupportsWireEffort('GLM-5.1')).toBe(true) + expect(resolveAppliedEffort('GLM-5.1', 'xhigh')).toBe('high') +}) + +test('provider override support context ignores ambient catalog metadata', async () => { + const { modelSupportsShimReasoningEffort } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: true, + routeId: 'custom-gateway', + catalogEntries: [ + { + id: 'gpt-5.4', + apiName: 'gpt-5.4', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'always-on', + wireFormat: 'none', + }, + }, + ], + }) + + expect(modelSupportsShimReasoningEffort( + 'gpt-5.4', + undefined, + undefined, + { routeId: 'openai', useRuntimeFallback: false }, + )).toBe(true) +}) +test('OpenAI shim reasoning request plan centralizes DeepSeek and Z.AI serialization', async () => { + const { resolveOpenAIShimReasoningRequestPlan } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + }) + + expect(resolveOpenAIShimReasoningRequestPlan({ + model: 'deepseek-v4-pro', + requestedEffort: 'xhigh', + requestThinkingType: 'enabled', + thinkingRequestFormat: 'deepseek-compatible', + })).toEqual({ + thinkingType: 'enabled', + reasoningEffort: 'max', + wireFormat: 'deepseek_compatible', + source: 'compat', + }) + + expect(resolveOpenAIShimReasoningRequestPlan({ + model: 'glm-5.2', + requestedEffort: 'xhigh', + thinkingRequestFormat: 'zai-compatible', + })).toEqual({ + thinkingType: 'enabled', + reasoningEffort: 'max', + wireFormat: 'zai_compatible', + source: 'compat', + }) + + expect(resolveOpenAIShimReasoningRequestPlan({ + model: 'GLM-5.1', + requestedEffort: 'high', + thinkingRequestFormat: 'zai-compatible', + })).toEqual({ + thinkingType: 'enabled', + reasoningEffort: undefined, + wireFormat: 'zai_compatible', + source: 'compat', + }) +}) + +test('explicit non-generic metadata wire formats stay non-controllable until planner support exists', async () => { + const { + modelSupportsEffort, + modelSupportsWireEffort, + resolveModelReasoningControl, + } = await importFreshEffortModule({ + provider: 'openai', + supportsCodexReasoningEffort: false, + routeId: 'custom-gateway', + catalogEntries: [ + { + id: 'custom-deepseek-model', + apiName: 'custom-deepseek-model', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['high', 'xhigh'], + wireFormat: 'deepseek_compatible', + }, + }, + ], + }) + + expect(resolveModelReasoningControl('custom-deepseek-model')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'metadata', + wireFormat: 'deepseek_compatible', + }) + expect(modelSupportsEffort('custom-deepseek-model')).toBe(false) + expect(modelSupportsWireEffort('custom-deepseek-model')).toBe(false) +}) diff --git a/src/utils/effort.ts b/src/utils/effort.ts index 9dbe35af7b..a11ce4220b 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -12,8 +12,10 @@ import { getModel, resolveActiveRouteIdFromEnv, } from '../integrations/index.js' +import { resolveOpenAIShimRuntimeContext } from '../integrations/runtimeMetadata.js' import type { CapabilityFlags, + OpenAIShimTransportConfig, ReasoningControlMetadata, ReasoningWireFormat, } from '../integrations/descriptors.js' @@ -47,10 +49,26 @@ export type ReasoningControlResolution = { levels: EffortLevel[] defaultLevel?: EffortValue wireFormat?: ReasoningWireFormat - source: 'metadata' | 'capability' | 'legacy' | 'none' + source: 'metadata' | 'capability' | 'compat' | 'legacy' | 'none' +} + +export type OpenAIShimThinkingRequestFormat = + NonNullable + +export type OpenAIShimReasoningRequestPlan = { + thinkingType?: 'enabled' | 'disabled' + reasoningEffort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max' + wireFormat?: ReasoningWireFormat + source: 'metadata' | 'legacy' | 'compat' | 'none' +} + +type OpenAIShimReasoningSupportContext = { + routeId?: string | null + useRuntimeFallback?: boolean } const DEFAULT_REASONING_LEVELS: EffortLevel[] = ['low', 'medium', 'high'] +const OPENAI_SHIM_COMPAT_LEVELS: EffortLevel[] = ['low', 'medium', 'high', 'xhigh'] function isSupportedEffortLevel(level: string): level is EffortLevel { return (EFFORT_LEVELS as readonly string[]).includes(level) @@ -78,18 +96,180 @@ function normalizeReasoningDefaultLevel( function metadataWireFormatSupportsEffort( wireFormat: ReasoningWireFormat | undefined, ): boolean { - // PR 1 only enables the existing OpenAI-compatible effort serializer. - // Other wire formats are typed for follow-up provider PRs, but must not - // silently flow through as top-level reasoning_effort before their request - // shapers are implemented. + // Explicit metadata is controllable only when the planner consumes that exact + // wire format directly. DeepSeek/Z.AI formats are currently enabled through + // temporary compatibility rules, not catalog metadata. return wireFormat === 'reasoning_effort' } -function resolveCatalogReasoningMetadata(model: string): { +function normalizedBaseModel(model: string | undefined): string { + return model?.trim().split('?', 1)[0]?.trim().toLowerCase() ?? '' +} + +function providerScopedModelSegments(model: string): string[] { + const segments = normalizedBaseModel(model) + .split('/') + .map(segment => segment.trim()) + .filter(Boolean) + const suffixes = segments + .slice(1) + .map((_, index) => segments.slice(index + 1).join('/')) + const accountQualifiedSuffixes = suffixes + .filter(suffix => /^[^/]+\/models\//.test(suffix)) + .map(suffix => `accounts/${suffix}`) + + return [...segments, ...suffixes, ...accountQualifiedSuffixes] +} + +function modelLooksDeepSeekCompatible(model: string): boolean { + return providerScopedModelSegments(model).some(segment => + segment.startsWith('deepseek'), + ) +} + +function modelLooksZaiCompatible(model: string): boolean { + const normalized = normalizedBaseModel(model) + return normalized.startsWith('glm-') || normalized.startsWith('zai-org/glm-') +} + +function supportsZaiReasoningEffort(model: string | undefined): boolean { + const normalized = normalizedBaseModel(model) + return normalized === 'glm-5.2' || normalized === 'zai-org/glm-5.2' +} + +function normalizeReasoningThinkingType( + value: string | undefined, +): 'enabled' | 'disabled' | undefined { + const normalized = value?.trim().toLowerCase() + if (normalized === 'disabled') { + return 'disabled' + } + if (normalized === 'enabled' || normalized === 'adaptive') { + return 'enabled' + } + return undefined +} + +function normalizeDeepSeekReasoningEffort( + effort: 'low' | 'medium' | 'high' | 'xhigh', +): 'high' | 'max' { + return effort === 'xhigh' ? 'max' : 'high' +} + +function normalizeZaiReasoningEffort( + effort: 'low' | 'medium' | 'high' | 'xhigh', +): 'high' | 'max' { + return effort === 'xhigh' ? 'max' : 'high' +} + +function resolveCompatibilityWireFormat( + model: string, + thinkingRequestFormat?: OpenAIShimThinkingRequestFormat, + routeIdOverride?: string | null, + useRuntimeFallback = true, +): ReasoningWireFormat | undefined { + if (thinkingRequestFormat === 'deepseek-compatible') { + return 'deepseek_compatible' + } + if (thinkingRequestFormat === 'zai-compatible') { + return 'zai_compatible' + } + if (thinkingRequestFormat === 'none') { + return undefined + } + + const routeId = routeIdOverride !== undefined + ? routeIdOverride + : useRuntimeFallback + ? resolveActiveRouteIdFromEnv(process.env) + : undefined + if (!routeId || routeId === 'anthropic' || routeId === 'openai') { + return undefined + } + if (modelLooksDeepSeekCompatible(model)) { + return 'deepseek_compatible' + } + if (routeId === 'zai' && modelLooksZaiCompatible(model)) { + return 'zai_compatible' + } + return undefined +} + +function resolveCompatibilityReasoningControl( + model: string, + thinkingRequestFormat?: OpenAIShimThinkingRequestFormat, + removeBodyFields?: string[], + context?: OpenAIShimReasoningSupportContext, +): ReasoningControlResolution | undefined { + const useRuntimeFallback = context?.useRuntimeFallback ?? true + const runtimeShimConfig = useRuntimeFallback && thinkingRequestFormat === undefined && removeBodyFields === undefined + ? resolveOpenAIShimRuntimeContext({ + processEnv: process.env, + model, + }).openaiShimConfig + : undefined + const resolvedThinkingRequestFormat = + thinkingRequestFormat ?? runtimeShimConfig?.thinkingRequestFormat + const resolvedRemoveBodyFields = + removeBodyFields ?? runtimeShimConfig?.removeBodyFields + const wireFormat = resolveCompatibilityWireFormat( + model, + resolvedThinkingRequestFormat, + context?.routeId, + useRuntimeFallback, + ) + if (!wireFormat) { + return undefined + } + + if (wireFormat === 'deepseek_compatible') { + if (resolvedRemoveBodyFields?.includes('reasoning_effort')) { + return undefined + } + return { + supportsReasoning: true, + controllable: true, + mode: 'levels', + levels: [...OPENAI_SHIM_COMPAT_LEVELS], + defaultLevel: undefined, + wireFormat, + source: 'compat', + } + } + + if (wireFormat === 'zai_compatible') { + const reasoningEffortStripped = + resolvedRemoveBodyFields?.includes('reasoning_effort') === true + const levels: EffortLevel[] = supportsZaiReasoningEffort(model) && !reasoningEffortStripped + ? ['high', 'xhigh'] + : ['high'] + return { + supportsReasoning: true, + controllable: true, + mode: 'levels', + levels, + defaultLevel: undefined, + wireFormat, + source: 'compat', + } + } + + return undefined +} + +function resolveCatalogReasoningMetadata( + model: string, + routeIdOverride?: string | null, + useRuntimeFallback = true, +): { capabilities?: CapabilityFlags reasoning?: ReasoningControlMetadata } | undefined { - const routeId = resolveActiveRouteIdFromEnv(process.env) + const routeId = routeIdOverride !== undefined + ? routeIdOverride + : useRuntimeFallback + ? resolveActiveRouteIdFromEnv(process.env) + : undefined if (!routeId || routeId === 'anthropic') { return undefined } @@ -116,8 +296,14 @@ function resolveCatalogReasoningMetadata(model: string): { function resolveMetadataReasoningControl( model: string, + routeIdOverride?: string | null, + useRuntimeFallback = true, ): ReasoningControlResolution | undefined { - const metadata = resolveCatalogReasoningMetadata(model) + const metadata = resolveCatalogReasoningMetadata( + model, + routeIdOverride, + useRuntimeFallback, + ) if (!metadata) { return undefined } @@ -224,6 +410,15 @@ export function resolveModelReasoningControl( model: string, ): ReasoningControlResolution { const metadata = resolveMetadataReasoningControl(model) + if (metadata?.source === 'metadata') { + return metadata + } + + const compatibility = resolveCompatibilityReasoningControl(model) + if (compatibility) { + return compatibility + } + if (metadata) { return metadata } @@ -243,7 +438,12 @@ export function modelSupportsEffort(model: string): boolean { return resolveModelReasoningControl(model).controllable } -export function modelSupportsWireEffort(model: string): boolean { +export function modelSupportsShimReasoningEffort( + model: string, + thinkingRequestFormat?: OpenAIShimThinkingRequestFormat, + removeBodyFields?: string[], + context?: OpenAIShimReasoningSupportContext, +): boolean { if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { return true } @@ -251,9 +451,112 @@ export function modelSupportsWireEffort(model: string): boolean { if (supported3P !== undefined) { return supported3P } - const control = resolveModelReasoningControl(model) + + const metadata = resolveMetadataReasoningControl( + model, + context?.routeId, + context?.useRuntimeFallback ?? true, + ) + if (metadata?.source === 'metadata') { + return Boolean(metadata.controllable && metadataWireFormatSupportsEffort(metadata.wireFormat)) + } + + const compatibility = resolveCompatibilityReasoningControl( + model, + thinkingRequestFormat, + removeBodyFields, + context, + ) + if (compatibility) { + return compatibility.controllable + } + + if ( + context?.routeId && + (context.routeId === 'openai' || context.routeId === 'codex') && + !removeBodyFields?.includes('reasoning_effort') + ) { + return supportsCodexReasoningEffort(model) + } + + if (context?.useRuntimeFallback === false) { + return false + } + + const control = metadata ?? resolveLegacyReasoningControl(model) return Boolean(control.controllable && metadataWireFormatSupportsEffort(control.wireFormat)) } + +export function modelSupportsWireEffort(model: string): boolean { + if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { + return true + } + const supported3P = get3PModelCapabilityOverride(model, 'effort') + if (supported3P !== undefined) { + return supported3P + } + return modelSupportsShimReasoningEffort(model) +} + +export function resolveOpenAIShimReasoningRequestPlan(options: { + model: string + requestedEffort?: OpenAIEffortLevel + requestThinkingType?: string + defaultThinkingType?: string + thinkingRequestFormat?: OpenAIShimThinkingRequestFormat +}): OpenAIShimReasoningRequestPlan { + const wireFormat = resolveCompatibilityWireFormat( + options.model, + options.thinkingRequestFormat, + ) + const requestedThinkingType = normalizeReasoningThinkingType( + options.requestThinkingType, + ) + const defaultThinkingType = normalizeReasoningThinkingType( + options.defaultThinkingType, + ) + + if (wireFormat === 'deepseek_compatible') { + const thinkingType = requestedThinkingType + const reasoningEffort = thinkingType === 'enabled' && options.requestedEffort + ? normalizeDeepSeekReasoningEffort(options.requestedEffort) + : undefined + return { + thinkingType, + reasoningEffort, + wireFormat, + source: 'compat', + } + } + + if (wireFormat === 'zai_compatible') { + const thinkingType = requestedThinkingType ?? defaultThinkingType + if (thinkingType === 'disabled') { + return { + thinkingType: 'disabled', + wireFormat, + source: 'compat', + } + } + + const shouldEnableThinking = thinkingType === 'enabled' || options.requestedEffort !== undefined + const reasoningEffort = options.requestedEffort && supportsZaiReasoningEffort(options.model) + ? normalizeZaiReasoningEffort(options.requestedEffort) + : undefined + return { + thinkingType: shouldEnableThinking ? 'enabled' : undefined, + reasoningEffort, + wireFormat, + source: 'compat', + } + } + + return { + reasoningEffort: options.requestedEffort, + wireFormat: options.requestedEffort ? 'reasoning_effort' : undefined, + source: options.requestedEffort ? 'legacy' : 'none', + } +} // @[MODEL LAUNCH]: Add the new model to the allowlist if it supports 'max' effort. // Per API docs, 'max' is Opus 4.6 only for public models — other models return an error. function legacyModelSupportsMaxEffort(model: string): boolean { @@ -341,7 +644,7 @@ function getLegacyAvailableEffortLevels(model: string): EffortLevel[] { export function modelSupportsMaxEffort(model: string): boolean { const control = resolveModelReasoningControl(model) - if (control.source === 'metadata' || control.source === 'capability') { + if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return control.levels.includes('max') } return legacyModelSupportsMaxEffort(model) @@ -349,7 +652,7 @@ export function modelSupportsMaxEffort(model: string): boolean { export function modelSupportsXHighEffort(model: string): boolean { const control = resolveModelReasoningControl(model) - if (control.source === 'metadata' || control.source === 'capability') { + if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return control.levels.includes('xhigh') } return legacyModelSupportsXHighEffort(model) @@ -357,7 +660,7 @@ export function modelSupportsXHighEffort(model: string): boolean { export function getAvailableEffortLevels(model: string): EffortLevel[] { const control = resolveModelReasoningControl(model) - if (control.source === 'metadata' || control.source === 'capability') { + if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return [...control.levels] } return getLegacyAvailableEffortLevels(model) @@ -481,6 +784,15 @@ export function resolveAppliedEffort( const resolved = envOverride ?? appStateEffortValue ?? getDefaultEffortForModel(model) + const control = resolveModelReasoningControl(model) + if ( + typeof resolved === 'string' && + (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') && + control.levels.length > 0 && + !control.levels.includes(resolved) + ) { + return control.levels.includes('high') ? 'high' : (control.defaultLevel ?? control.levels[0]) + } // API rejects 'max' on non-Opus-4.6 Anthropic models — downgrade to 'high'. // OpenAI/Codex models use 'max' as the standard form of 'xhigh'; the client // shim converts it back to 'xhigh' on the wire, so don't clamp it here. @@ -553,7 +865,7 @@ export function getDefaultEffortForModel( model: string, ): EffortValue | undefined { const control = resolveModelReasoningControl(model) - if (control.source === 'metadata' || control.source === 'capability') { + if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return control.defaultLevel } return getLegacyDefaultEffortForModel(model) From 538750636bac47e0a5291bcc6697ee3f97b8e619 Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 15:03:56 -0700 Subject: [PATCH 3/9] Fix Responses API reasoning effort shape Send reasoning effort on OpenAI-compatible Responses requests using the nested reasoning object expected by the endpoint instead of flat reasoning_effort/reasoning_summary fields. Keep chat_completions behavior unchanged so OpenAI-compatible chat endpoints still receive top-level reasoning_effort. Add a regression test covering the Responses request body shape and verifying the flat fields are omitted. Validation: bun test --feature=UNATTENDED_RETRY src/services/api/openaiShim.test.ts; node .\\node_modules\\typescript\\bin\\tsc --noEmit; git diff --check. --- src/services/api/openaiShim.test.ts | 42 +++++++++++++++++++++++++++++ src/services/api/openaiShim.ts | 8 +++--- 2 files changed, 47 insertions(+), 3 deletions(-) diff --git a/src/services/api/openaiShim.test.ts b/src/services/api/openaiShim.test.ts index 23787ba856..d104088ddf 100644 --- a/src/services/api/openaiShim.test.ts +++ b/src/services/api/openaiShim.test.ts @@ -357,6 +357,48 @@ test('uses OpenAI-compatible responses endpoint when OPENAI_API_FORMAT=responses ]) }) +test('nests reasoning effort for OpenAI-compatible responses endpoint', async () => { + process.env.OPENAI_API_FORMAT = 'responses' + let capturedBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + capturedBody = JSON.parse(String(init?.body)) as Record + + return new Response( + JSON.stringify({ + id: 'resp-1', + model: 'gpt-5.4', + output: [ + { + type: 'message', + role: 'assistant', + content: [{ type: 'output_text', text: 'ok' }], + }, + ], + }), + { + headers: { + 'Content-Type': 'application/json', + }, + }, + ) + }) as unknown as FetchType + + const client = createOpenAIShimClient({ reasoningEffort: 'high' }) as OpenAIShimClient + + await client.beta.messages.create({ + model: 'gpt-5.4', + messages: [{ role: 'user', content: 'hello' }], + max_tokens: 64, + stream: false, + }) + + expect(capturedBody?.reasoning).toEqual({ effort: 'high', summary: 'auto' }) + expect(capturedBody?.include).toEqual(['reasoning.encrypted_content']) + expect(capturedBody).not.toHaveProperty('reasoning_effort') + expect(capturedBody).not.toHaveProperty('reasoning_summary') +}) + test('uses OpenAI-compatible responses endpoint with text chunk types when OPENAI_API_FORMAT=responses_compat', async () => { process.env.OPENAI_API_FORMAT = 'responses_compat' let capturedUrl = '' diff --git a/src/services/api/openaiShim.ts b/src/services/api/openaiShim.ts index 69c9d14c74..e9b047925a 100644 --- a/src/services/api/openaiShim.ts +++ b/src/services/api/openaiShim.ts @@ -2614,9 +2614,11 @@ class OpenAIShimMessages { if (params.temperature !== undefined) responsesBody.temperature = params.temperature if (params.top_p !== undefined) responsesBody.top_p = params.top_p - if (request.reasoning?.effort) { - responsesBody.reasoning_effort = request.reasoning.effort - responsesBody.reasoning_summary = 'auto' + if (reasoningRequestPlan.wireFormat === 'reasoning_effort' && reasoningRequestPlan.reasoningEffort) { + responsesBody.reasoning = { + effort: reasoningRequestPlan.reasoningEffort, + summary: 'auto', + } responsesBody.include = ['reasoning.encrypted_content'] } From 39badf11930d761cbd816f51c86f1b73173b4e7a Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 15:13:01 -0700 Subject: [PATCH 4/9] Document reasoning effort metadata rules Make the new reasoning-effort guide discoverable from the integrations overview and reading order. Update model, gateway, and vendor onboarding docs to explain that supportsReasoning is descriptive only and does not enable /effort request mutation without verified per-model reasoning metadata. Clarify that gateway and vendor catalogs must annotate reasoning controls per exact route/model rather than provider-wide. Validation: git diff --check. --- docs/integrations/how-to/add-gateway.md | 12 ++++++++++++ docs/integrations/how-to/add-model.md | 11 +++++++++++ docs/integrations/how-to/add-vendor.md | 12 ++++++++++++ docs/integrations/overview.md | 17 ++++++++++++++--- docs/integrations/reasoning-effort.md | 2 +- 5 files changed, 50 insertions(+), 4 deletions(-) diff --git a/docs/integrations/how-to/add-gateway.md b/docs/integrations/how-to/add-gateway.md index 560f37efc8..630975d139 100644 --- a/docs/integrations/how-to/add-gateway.md +++ b/docs/integrations/how-to/add-gateway.md @@ -55,6 +55,18 @@ Normal gateway examples should: The routing decision belongs to `transportConfig.kind`, not to `category`. +## Reasoning controls in mixed catalogs + +Gateway catalogs often mix models with different reasoning APIs. Keep +`capabilities.supportsReasoning` as descriptive capability metadata unless +the exact gateway route and model ID have been probed. + +Add `/effort`-controllable `reasoning` metadata per catalog entry, not at the +gateway level. If a gateway accepts one upstream model's `reasoning_effort` +but rejects another model's field, each entry must say so explicitly. See +`docs/integrations/reasoning-effort.md` before adding or changing reasoning +controls. + ## Generated loader and preset manifest Normal gateway onboarding is additive now: diff --git a/docs/integrations/how-to/add-model.md b/docs/integrations/how-to/add-model.md index 5e0895b61f..bff5bb505c 100644 --- a/docs/integrations/how-to/add-model.md +++ b/docs/integrations/how-to/add-model.md @@ -65,6 +65,17 @@ editing multiple shared model files. In the common path: one route; - let the route catalog continue to own the offered subset. +## Reasoning and `/effort` metadata + +Use `classification: ['reasoning']` and `capabilities.supportsReasoning` to +describe that a model is known to reason or think. Those fields do not by +themselves enable `/effort` request mutation. + +Only add `reasoning` metadata when the model's exact control surface has been +verified for the route that will call it, including accepted levels, rejected +levels, and any thinking-disable format. For the full checklist, see +`docs/integrations/reasoning-effort.md`. + ## When to add a brand descriptor Add or update a brand descriptor when: diff --git a/docs/integrations/how-to/add-vendor.md b/docs/integrations/how-to/add-vendor.md index 662353577d..6256202b42 100644 --- a/docs/integrations/how-to/add-vendor.md +++ b/docs/integrations/how-to/add-vendor.md @@ -251,6 +251,18 @@ context windows, output limits, and cross-route capability metadata in `src/integrations/models/`, then point catalog entries at those descriptors with `modelDescriptorId`. +## Reasoning controls + +For direct vendors, record reasoning controls on the exact catalog model +entry or shared model descriptor only after the vendor API has been probed. +`capabilities.supportsReasoning` means the model can reason; it does not +mean `/effort` should send `reasoning_effort` or any other control field. + +If the vendor catalog contains both controllable and non-controllable +reasoning models, annotate each model separately. See +`docs/integrations/reasoning-effort.md` for the metadata shape and audit +checklist. + ## OpenAI-compatible UI capability flags For OpenAI-compatible vendors, be explicit about the provider editor surface: diff --git a/docs/integrations/overview.md b/docs/integrations/overview.md index 8894bc8cb6..8e1b0ba798 100644 --- a/docs/integrations/overview.md +++ b/docs/integrations/overview.md @@ -24,6 +24,7 @@ docs/ integrations/ overview.md glossary.md + reasoning-effort.md how-to/ add-vendor.md add-gateway.md @@ -43,9 +44,11 @@ If you are onboarding to the integration system: 1. Read `docs/architecture/integrations.md` for the system boundaries. 2. Read `docs/integrations/glossary.md` for the shared vocabulary. -3. Use the how-to guides for the specific descriptor type you are adding. -4. Use `docs/integrations/reference-samples.md` once the architecture and the relevant how-to guide are clear. -5. Read `docs/integrations/common-pitfalls.md` before opening a docs or implementation PR for a new integration. +3. Read `docs/integrations/reasoning-effort.md` before marking models as + reasoning-capable or `/effort`-controllable. +4. Use the how-to guides for the specific descriptor type you are adding. +5. Use `docs/integrations/reference-samples.md` once the architecture and the relevant how-to guide are clear. +6. Read `docs/integrations/common-pitfalls.md` before opening a docs or implementation PR for a new integration. ## Core Rules @@ -115,6 +118,14 @@ routes. Fixed direct vendors usually set both to `false`; broad custom routes or gateways that intentionally accept user-supplied auth/header details set the relevant flag to `true`. +### Reasoning support is per model and per route + +`capabilities.supportsReasoning` is descriptive. It says the model is known to +reason or think, but it does not by itself authorize `/effort` to add request +fields. Only add `reasoning` metadata when the exact route/model request shape, +accepted levels, and disable behavior have been verified. See +`docs/integrations/reasoning-effort.md`. + ## Descriptor Authoring Pattern Normal descriptor files should: diff --git a/docs/integrations/reasoning-effort.md b/docs/integrations/reasoning-effort.md index d307f5a94c..a3f8a8a282 100644 --- a/docs/integrations/reasoning-effort.md +++ b/docs/integrations/reasoning-effort.md @@ -55,4 +55,4 @@ Before adding `reasoning` metadata for a model: 4. Confirm whether accepted parameters actually change behavior or are silent no-ops. 5. Add focused tests for the resolver and request serialization path. -Do not use `supportsReasoning: true` alone as evidence that `reasoning_effort` or any other effort field is accepted. \ No newline at end of file +Do not use `supportsReasoning: true` alone as evidence that `reasoning_effort` or any other effort field is accepted. From 462114593f8ce44694667be4eccc97d5fd9a4696 Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 15:42:27 -0700 Subject: [PATCH 5/9] Stabilize effort resolver tests Add an optional reasoning control context so effort tests can inject provider, catalog, model descriptor, and shim metadata without mocking process-global integration/provider modules. Update effort.codex tests to use the injected context and restore only the remaining local mocks, preventing mock leakage into later full-suite provider tests. Validation: bun run check; bun run test:provider; bun run test:provider-recommendation; bun run typecheck:type-tests; bun run integrations:check; python -m pytest -q python/tests; bun run security:pr-scan -- --base upstream/main --head HEAD. --- src/utils/effort.codex.test.ts | 86 +++++++++++------ src/utils/effort.ts | 167 +++++++++++++++++++++------------ 2 files changed, 162 insertions(+), 91 deletions(-) diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index 5b494706b1..1569d1e217 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -10,13 +10,17 @@ import { // downstream tests that load it via openaiShim/client/codexShim crash with // "Export named 'X' not found in module". import * as actualAuth from './auth.js' -import * as actualProviderConfig from '../services/api/providerConfig.js' import * as actualThinking from './thinking.js' import * as actualGrowthbook from 'src/services/analytics/growthbook.js' -import * as actualProviders from './model/providers.js' import * as actualModelSupportOverrides from './model/modelSupportOverrides.js' -import * as actualIntegrations from '../integrations/index.js' -import * as actualRuntimeMetadata from '../integrations/runtimeMetadata.js' + +function restoreMockedModulesToActual(): void { + mock.module('./model/modelSupportOverrides.js', () => actualModelSupportOverrides) + mock.module('./auth.js', () => actualAuth) + mock.module('./thinking.js', () => actualThinking) + mock.module('src/services/analytics/growthbook.js', () => actualGrowthbook) +} + beforeEach(async () => { await acquireSharedMutationLock('utils/effort.codex.test.ts') @@ -25,6 +29,7 @@ beforeEach(async () => { afterEach(() => { try { mock.restore() + restoreMockedModulesToActual() } finally { releaseSharedMutationLock() } @@ -38,34 +43,10 @@ async function importFreshEffortModule(options: { modelDescriptors?: Record openaiShimConfig?: any }) { - mock.module('./model/providers.js', () => ({ - ...actualProviders, - getAPIProvider: () => options.provider, - })) mock.module('./model/modelSupportOverrides.js', () => ({ ...actualModelSupportOverrides, get3PModelCapabilityOverride: () => undefined, })) - mock.module('../services/api/providerConfig.js', () => ({ - ...actualProviderConfig, - supportsCodexReasoningEffort: () => options.supportsCodexReasoningEffort, - })) - mock.module('../integrations/index.js', () => ({ - ...actualIntegrations, - resolveActiveRouteIdFromEnv: () => options.routeId, - getCatalogEntriesForRoute: (routeId: string) => - routeId === options.routeId ? (options.catalogEntries ?? []) : [], - getModel: (id: string) => options.modelDescriptors?.[id], - })) - mock.module('../integrations/runtimeMetadata.js', () => ({ - ...actualRuntimeMetadata, - resolveOpenAIShimRuntimeContext: () => ({ - routeId: options.routeId ?? null, - descriptor: null, - catalogEntry: null, - openaiShimConfig: options.openaiShimConfig ?? {}, - }), - })) mock.module('./auth.js', () => ({ ...actualAuth, isProSubscriber: () => false, @@ -82,7 +63,54 @@ async function importFreshEffortModule(options: { fallback, })) - return import(`./effort.js?ts=${Date.now()}-${Math.random()}`) + const effort = await import(`./effort.js?ts=${Date.now()}-${Math.random()}`) + const reasoningContext = ( + options.provider !== undefined || + options.supportsCodexReasoningEffort !== undefined || + options.routeId !== undefined || + options.catalogEntries !== undefined || + options.modelDescriptors !== undefined || + options.openaiShimConfig !== undefined + ) + ? { + apiProvider: options.provider, + supportsCodexReasoningEffort: options.supportsCodexReasoningEffort, + routeId: options.routeId, + catalogEntries: options.catalogEntries, + modelDescriptors: options.modelDescriptors, + openaiShimConfig: options.openaiShimConfig, + } + : undefined + + return { + ...effort, + resolveModelReasoningControl: (model: string) => + effort.resolveModelReasoningControl(model, reasoningContext), + modelSupportsEffort: (model: string) => + effort.modelSupportsEffort(model, reasoningContext), + modelSupportsWireEffort: (model: string) => + effort.modelSupportsWireEffort(model, reasoningContext), + getAvailableEffortLevels: (model: string) => + effort.getAvailableEffortLevels(model, reasoningContext), + modelUsesOpenAIEffort: (model: string) => + effort.modelUsesOpenAIEffort(model, reasoningContext), + getDefaultEffortForModel: (model: string) => + effort.getDefaultEffortForModel(model, reasoningContext), + resolveAppliedEffort: (model: string, appStateEffortValue: unknown) => + effort.resolveAppliedEffort(model, appStateEffortValue, reasoningContext), + modelSupportsShimReasoningEffort: ( + model: string, + thinkingRequestFormat?: unknown, + removeBodyFields?: string[], + context?: unknown, + ) => + effort.modelSupportsShimReasoningEffort( + model, + thinkingRequestFormat, + removeBodyFields, + context ?? reasoningContext, + ), + } } test('gpt-5.4 on the ChatGPT Codex backend supports effort selection', async () => { diff --git a/src/utils/effort.ts b/src/utils/effort.ts index a11ce4220b..a8958322ee 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -15,6 +15,8 @@ import { import { resolveOpenAIShimRuntimeContext } from '../integrations/runtimeMetadata.js' import type { CapabilityFlags, + ModelCatalogEntry, + ModelDescriptor, OpenAIShimTransportConfig, ReasoningControlMetadata, ReasoningWireFormat, @@ -67,9 +69,34 @@ type OpenAIShimReasoningSupportContext = { useRuntimeFallback?: boolean } +export type ReasoningControlContext = OpenAIShimReasoningSupportContext & { + apiProvider?: ReturnType + supportsCodexReasoningEffort?: boolean | ((model: string) => boolean) + catalogEntries?: readonly ModelCatalogEntry[] + modelDescriptors?: Readonly>> + openaiShimConfig?: Partial +} + const DEFAULT_REASONING_LEVELS: EffortLevel[] = ['low', 'medium', 'high'] const OPENAI_SHIM_COMPAT_LEVELS: EffortLevel[] = ['low', 'medium', 'high', 'xhigh'] +function getReasoningApiProvider( + context?: ReasoningControlContext, +): ReturnType { + return context?.apiProvider ?? getAPIProvider() +} + +function modelSupportsCodexReasoningEffort( + model: string, + context?: ReasoningControlContext, +): boolean { + const override = context?.supportsCodexReasoningEffort + if (typeof override === 'function') { + return override(model) + } + return override ?? supportsCodexReasoningEffort(model) +} + function isSupportedEffortLevel(level: string): level is EffortLevel { return (EFFORT_LEVELS as readonly string[]).includes(level) } @@ -199,15 +226,15 @@ function resolveCompatibilityReasoningControl( model: string, thinkingRequestFormat?: OpenAIShimThinkingRequestFormat, removeBodyFields?: string[], - context?: OpenAIShimReasoningSupportContext, + context?: ReasoningControlContext, ): ReasoningControlResolution | undefined { const useRuntimeFallback = context?.useRuntimeFallback ?? true - const runtimeShimConfig = useRuntimeFallback && thinkingRequestFormat === undefined && removeBodyFields === undefined + const runtimeShimConfig = context?.openaiShimConfig ?? (useRuntimeFallback && thinkingRequestFormat === undefined && removeBodyFields === undefined ? resolveOpenAIShimRuntimeContext({ processEnv: process.env, model, }).openaiShimConfig - : undefined + : undefined) const resolvedThinkingRequestFormat = thinkingRequestFormat ?? runtimeShimConfig?.thinkingRequestFormat const resolvedRemoveBodyFields = @@ -259,23 +286,23 @@ function resolveCompatibilityReasoningControl( function resolveCatalogReasoningMetadata( model: string, - routeIdOverride?: string | null, - useRuntimeFallback = true, + context?: ReasoningControlContext, ): { capabilities?: CapabilityFlags reasoning?: ReasoningControlMetadata } | undefined { - const routeId = routeIdOverride !== undefined - ? routeIdOverride - : useRuntimeFallback - ? resolveActiveRouteIdFromEnv(process.env) - : undefined + const routeId = context?.routeId !== undefined + ? context.routeId + : context?.useRuntimeFallback === false + ? undefined + : resolveActiveRouteIdFromEnv(process.env) if (!routeId || routeId === 'anthropic') { return undefined } const normalizedModel = model.trim().split('?', 1)[0]!.trim().toLowerCase() - const entry = getCatalogEntriesForRoute(routeId).find(catalogEntry => + const entries = context?.catalogEntries ?? getCatalogEntriesForRoute(routeId) + const entry = entries.find(catalogEntry => catalogEntry.apiName.trim().toLowerCase() === normalizedModel || catalogEntry.id.trim().toLowerCase() === normalizedModel, ) @@ -285,7 +312,7 @@ function resolveCatalogReasoningMetadata( } const descriptor = entry.modelDescriptorId - ? getModel(entry.modelDescriptorId) + ? context?.modelDescriptors?.[entry.modelDescriptorId] ?? getModel(entry.modelDescriptorId) : undefined return { @@ -296,13 +323,11 @@ function resolveCatalogReasoningMetadata( function resolveMetadataReasoningControl( model: string, - routeIdOverride?: string | null, - useRuntimeFallback = true, + context?: ReasoningControlContext, ): ReasoningControlResolution | undefined { const metadata = resolveCatalogReasoningMetadata( model, - routeIdOverride, - useRuntimeFallback, + context, ) if (!metadata) { return undefined @@ -342,7 +367,10 @@ function resolveMetadataReasoningControl( } } -function legacyModelSupportsEffort(model: string): boolean { +function legacyModelSupportsEffort( + model: string, + context?: ReasoningControlContext, +): boolean { const m = model.toLowerCase() if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { return true @@ -351,7 +379,7 @@ function legacyModelSupportsEffort(model: string): boolean { if (supported3P !== undefined) { return supported3P } - if (modelUsesOpenAIEffort(model) && supportsCodexReasoningEffort(model)) { + if (modelUsesOpenAIEffort(model, context) && modelSupportsCodexReasoningEffort(model, context)) { return true } // Claude 4 models that support effort. Mirrors the Anthropic /messages @@ -382,11 +410,14 @@ function legacyModelSupportsEffort(model: string): boolean { // Default to true for unknown model strings on 1P. // Do not default to true for 3P as they have different formats for their // model strings (ex. anthropics/claude-code#30795) - return getAPIProvider() === 'firstParty' + return getReasoningApiProvider(context) === 'firstParty' } -function resolveLegacyReasoningControl(model: string): ReasoningControlResolution { - if (!legacyModelSupportsEffort(model)) { +function resolveLegacyReasoningControl( + model: string, + context?: ReasoningControlContext, +): ReasoningControlResolution { + if (!legacyModelSupportsEffort(model, context)) { return { supportsReasoning: false, controllable: false, @@ -399,8 +430,8 @@ function resolveLegacyReasoningControl(model: string): ReasoningControlResolutio supportsReasoning: true, controllable: true, mode: 'levels', - levels: getLegacyAvailableEffortLevels(model), - defaultLevel: getLegacyDefaultEffortForModel(model), + levels: getLegacyAvailableEffortLevels(model, context), + defaultLevel: getLegacyDefaultEffortForModel(model, context), wireFormat: 'reasoning_effort', source: 'legacy', } @@ -408,13 +439,14 @@ function resolveLegacyReasoningControl(model: string): ReasoningControlResolutio export function resolveModelReasoningControl( model: string, + context?: ReasoningControlContext, ): ReasoningControlResolution { - const metadata = resolveMetadataReasoningControl(model) + const metadata = resolveMetadataReasoningControl(model, context) if (metadata?.source === 'metadata') { return metadata } - const compatibility = resolveCompatibilityReasoningControl(model) + const compatibility = resolveCompatibilityReasoningControl(model, undefined, undefined, context) if (compatibility) { return compatibility } @@ -423,11 +455,11 @@ export function resolveModelReasoningControl( return metadata } - return resolveLegacyReasoningControl(model) + return resolveLegacyReasoningControl(model, context) } // @[MODEL LAUNCH]: Add the new model to the allowlist if it supports the effort parameter. -export function modelSupportsEffort(model: string): boolean { +export function modelSupportsEffort(model: string, context?: ReasoningControlContext): boolean { if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { return true } @@ -435,14 +467,14 @@ export function modelSupportsEffort(model: string): boolean { if (supported3P !== undefined) { return supported3P } - return resolveModelReasoningControl(model).controllable + return resolveModelReasoningControl(model, context).controllable } export function modelSupportsShimReasoningEffort( model: string, thinkingRequestFormat?: OpenAIShimThinkingRequestFormat, removeBodyFields?: string[], - context?: OpenAIShimReasoningSupportContext, + context?: ReasoningControlContext, ): boolean { if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { return true @@ -454,8 +486,7 @@ export function modelSupportsShimReasoningEffort( const metadata = resolveMetadataReasoningControl( model, - context?.routeId, - context?.useRuntimeFallback ?? true, + context, ) if (metadata?.source === 'metadata') { return Boolean(metadata.controllable && metadataWireFormatSupportsEffort(metadata.wireFormat)) @@ -476,18 +507,18 @@ export function modelSupportsShimReasoningEffort( (context.routeId === 'openai' || context.routeId === 'codex') && !removeBodyFields?.includes('reasoning_effort') ) { - return supportsCodexReasoningEffort(model) + return modelSupportsCodexReasoningEffort(model, context) } if (context?.useRuntimeFallback === false) { return false } - const control = metadata ?? resolveLegacyReasoningControl(model) + const control = metadata ?? resolveLegacyReasoningControl(model, context) return Boolean(control.controllable && metadataWireFormatSupportsEffort(control.wireFormat)) } -export function modelSupportsWireEffort(model: string): boolean { +export function modelSupportsWireEffort(model: string, context?: ReasoningControlContext): boolean { if (isEnvTruthy(process.env.CLAUDE_CODE_ALWAYS_ENABLE_EFFORT)) { return true } @@ -495,7 +526,7 @@ export function modelSupportsWireEffort(model: string): boolean { if (supported3P !== undefined) { return supported3P } - return modelSupportsShimReasoningEffort(model) + return modelSupportsShimReasoningEffort(model, undefined, undefined, context) } export function resolveOpenAIShimReasoningRequestPlan(options: { @@ -576,15 +607,18 @@ function legacyModelSupportsMaxEffort(model: string): boolean { // @[MODEL LAUNCH]: Add the new model to the allowlist if it supports 'xhigh' effort. // xhigh is reserved for OpenAI/Codex models and OpenCode Claude opus 4-7 / 4-8. // All other effort-supporting models reject xhigh at the API. -function legacyModelSupportsXHighEffort(model: string): boolean { - if (!legacyModelSupportsEffort(model)) { +function legacyModelSupportsXHighEffort( + model: string, + context?: ReasoningControlContext, +): boolean { + if (!legacyModelSupportsEffort(model, context)) { return false } const supported3P = get3PModelCapabilityOverride(model, 'xhigh_effort') if (supported3P !== undefined) { return supported3P } - if (modelUsesOpenAIEffort(model)) { + if (modelUsesOpenAIEffort(model, context)) { return true } if (model.toLowerCase().includes('opus-4-7') || model.toLowerCase().includes('opus-4-8')) { @@ -601,8 +635,11 @@ export function isOpenAIEffortLevel(value: string): value is OpenAIEffortLevel { return (OPENAI_EFFORT_LEVELS as readonly string[]).includes(value) } -export function modelUsesOpenAIEffort(model: string): boolean { - const provider = getAPIProvider() +export function modelUsesOpenAIEffort( + model: string, + context?: ReasoningControlContext, +): boolean { + const provider = getReasoningApiProvider(context) if (provider !== 'openai' && provider !== 'codex') { return false } @@ -616,8 +653,11 @@ export function modelUsesOpenAIEffort(model: string): boolean { return true } -function getLegacyAvailableEffortLevels(model: string): EffortLevel[] { - if (!legacyModelSupportsEffort(model)) { +function getLegacyAvailableEffortLevels( + model: string, + context?: ReasoningControlContext, +): EffortLevel[] { + if (!legacyModelSupportsEffort(model, context)) { return [] } // OpenCode Claude and Gemini models use /messages or /models/gemini-* @@ -628,12 +668,12 @@ function getLegacyAvailableEffortLevels(model: string): EffortLevel[] { m.includes('claude-opus-4') || m.includes('claude-sonnet-4') || m.includes('opus-4') || m.includes('sonnet-4') || m.includes('gemini-3') - ) && getAPIProvider() === 'openai' - if (modelUsesOpenAIEffort(model) && !isOpenCodeNativeFormat) { + ) && getReasoningApiProvider(context) === 'openai' + if (modelUsesOpenAIEffort(model, context) && !isOpenCodeNativeFormat) { return [...OPENAI_EFFORT_LEVELS] as EffortLevel[] } const levels: EffortLevel[] = ['low', 'medium', 'high'] - if (legacyModelSupportsXHighEffort(model)) { + if (legacyModelSupportsXHighEffort(model, context)) { levels.push('xhigh') } if (legacyModelSupportsMaxEffort(model)) { @@ -642,28 +682,28 @@ function getLegacyAvailableEffortLevels(model: string): EffortLevel[] { return levels } -export function modelSupportsMaxEffort(model: string): boolean { - const control = resolveModelReasoningControl(model) +export function modelSupportsMaxEffort(model: string, context?: ReasoningControlContext): boolean { + const control = resolveModelReasoningControl(model, context) if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return control.levels.includes('max') } return legacyModelSupportsMaxEffort(model) } -export function modelSupportsXHighEffort(model: string): boolean { - const control = resolveModelReasoningControl(model) +export function modelSupportsXHighEffort(model: string, context?: ReasoningControlContext): boolean { + const control = resolveModelReasoningControl(model, context) if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return control.levels.includes('xhigh') } - return legacyModelSupportsXHighEffort(model) + return legacyModelSupportsXHighEffort(model, context) } -export function getAvailableEffortLevels(model: string): EffortLevel[] { - const control = resolveModelReasoningControl(model) +export function getAvailableEffortLevels(model: string, context?: ReasoningControlContext): EffortLevel[] { + const control = resolveModelReasoningControl(model, context) if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return [...control.levels] } - return getLegacyAvailableEffortLevels(model) + return getLegacyAvailableEffortLevels(model, context) } export function getEffortLevelLabel(level: EffortLevel | OpenAIEffortLevel): string { if (level === 'xhigh') return 'Extra High' @@ -773,18 +813,19 @@ export function getEffortEnvOverride(): EffortValue | null | undefined { export function resolveAppliedEffort( model: string, appStateEffortValue: EffortValue | undefined, + context?: ReasoningControlContext, ): EffortValue | undefined { const envOverride = getEffortEnvOverride() if (envOverride === null) { return undefined } - if (!modelSupportsEffort(model)) { + if (!modelSupportsEffort(model, context)) { return undefined } const resolved = - envOverride ?? appStateEffortValue ?? getDefaultEffortForModel(model) - const control = resolveModelReasoningControl(model) + envOverride ?? appStateEffortValue ?? getDefaultEffortForModel(model, context) + const control = resolveModelReasoningControl(model, context) if ( typeof resolved === 'string' && (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') && @@ -798,15 +839,15 @@ export function resolveAppliedEffort( // shim converts it back to 'xhigh' on the wire, so don't clamp it here. if ( resolved === 'max' && - !modelSupportsMaxEffort(model) && - !modelUsesOpenAIEffort(model) + !modelSupportsMaxEffort(model, context) && + !modelUsesOpenAIEffort(model, context) ) { return 'high' } // xhigh is reserved for OpenAI/Codex models and OpenCode opus-4-7/4-8. // For all other models, downgrade to 'high' so a stale persisted setting // doesn't surface as an API error. - if (resolved === 'xhigh' && !modelSupportsXHighEffort(model)) { + if (resolved === 'xhigh' && !modelSupportsXHighEffort(model, context)) { return 'high' } return resolved @@ -863,12 +904,13 @@ export function convertEffortValueToLevel(value: EffortValue): EffortLevel { export function getDefaultEffortForModel( model: string, + context?: ReasoningControlContext, ): EffortValue | undefined { - const control = resolveModelReasoningControl(model) + const control = resolveModelReasoningControl(model, context) if (control.source === 'metadata' || control.source === 'capability' || control.source === 'compat') { return control.defaultLevel } - return getLegacyDefaultEffortForModel(model) + return getLegacyDefaultEffortForModel(model, context) } /** * Get user-facing description for effort levels @@ -935,6 +977,7 @@ export function getOpusDefaultEffortConfig(): OpusDefaultEffortConfig { // @[MODEL LAUNCH]: Update the default effort levels for new models function getLegacyDefaultEffortForModel( model: string, + context?: ReasoningControlContext, ): EffortValue | undefined { if (process.env.USER_TYPE === 'ant') { const config = getAntModelOverrideConfig() @@ -976,7 +1019,7 @@ function getLegacyDefaultEffortForModel( } // When ultrathink feature is on, default effort to medium (ultrathink bumps to high) - if (isUltrathinkEnabled() && legacyModelSupportsEffort(model)) { + if (isUltrathinkEnabled() && legacyModelSupportsEffort(model, context)) { return 'medium' } From f4aef7038c0a6575a1e2bc248b7aa74a91b44029 Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 16:09:54 -0700 Subject: [PATCH 6/9] Address effort PR review findings Load integration registry before catalog reasoning lookup, isolate provider override route resolution from ambient routes, and carry explicit compat reasoning metadata into the OpenAI shim request planner. Clarify reasoning metadata documentation and add focused regression coverage for compat metadata and provider override route preference. --- docs/integrations/reasoning-effort.md | 7 ++-- src/integrations/runtimeMetadata.test.ts | 20 +++++++++ src/integrations/runtimeMetadata.ts | 9 ++-- src/services/api/openaiShim.ts | 13 +++++- src/utils/effort.codex.test.ts | 23 +++++++--- src/utils/effort.ts | 53 ++++++++++++++++++------ 6 files changed, 100 insertions(+), 25 deletions(-) diff --git a/docs/integrations/reasoning-effort.md b/docs/integrations/reasoning-effort.md index a3f8a8a282..a8429a9521 100644 --- a/docs/integrations/reasoning-effort.md +++ b/docs/integrations/reasoning-effort.md @@ -11,7 +11,8 @@ OpenClaude treats reasoning support as a per-model capability. Provider and gate ```ts reasoning: { mode: 'levels' | 'toggle' | 'always-on' - levels?: ['low', 'medium', 'high', 'xhigh', 'max'] + // Any supported subset for this exact model, for example ['high', 'xhigh']. + levels?: ReasoningEffortLevel[] defaultLevel?: 'low' | 'medium' | 'high' | 'xhigh' | 'max' wireFormat?: | 'reasoning_effort' @@ -31,11 +32,11 @@ The `/effort` resolver is intentionally conservative: 1. Explicit per-model `reasoning` metadata wins. 2. Existing hardcoded legacy effort support remains unchanged. 3. `supportsReasoning: true` without `reasoning` metadata is treated as reasoning-capable but not controllable. -4. Unknown models do not receive new reasoning request fields. +4. Truly unknown models do not receive new reasoning request fields. This means existing OpenAI, Codex, Claude, Gemini, and configured 3P override behavior remains active, while catalogs can safely mark models with `supportsReasoning` before their exact request shape has been audited. -A temporary compatibility layer also preserves verified request shaping that existed before per-model `reasoning` metadata. For example, DeepSeek-compatible routes can still map `/effort xhigh` to provider `reasoning_effort: "max"`, and Z.AI GLM routes can still map supported controls through their `thinking` request shape. These compatibility rules are intentionally centralized in the effort resolver so they can be removed as catalogs gain explicit `reasoning` metadata. +A temporary compatibility layer also preserves verified request shaping that existed before per-model `reasoning` metadata. For example, DeepSeek-compatible routes can still map `/effort xhigh` to provider `reasoning_effort: "max"`, and Z.AI GLM routes can still map supported controls through their `thinking` request shape. Those compatibility rules also cover matching uncataloged DeepSeek/Z.AI route traffic, so the unknown-model rule only applies after explicit metadata and compatibility resolution both fail. These rules are intentionally centralized in the effort resolver so they can be removed as catalogs gain explicit `reasoning` metadata. ## Provider and Gateway Rules diff --git a/src/integrations/runtimeMetadata.test.ts b/src/integrations/runtimeMetadata.test.ts index 189af06bf9..4a2cc01db3 100644 --- a/src/integrations/runtimeMetadata.test.ts +++ b/src/integrations/runtimeMetadata.test.ts @@ -184,6 +184,26 @@ describe('resolveOpenAIShimRuntimeContext - Z.AI GLM-5.2', () => { }) }) +describe('resolveOpenAIShimRuntimeContext - provider override route preference', () => { + it('does not inherit ambient route config when the preferred base URL is unrecognized', () => { + const result = resolveOpenAIShimRuntimeContext({ + model: 'gpt-4o', + baseUrl: 'https://custom.example.test/v1', + preferBaseUrlRoute: true, + processEnv: { + CLAUDE_CODE_USE_OPENAI: '1', + OPENAI_BASE_URL: 'https://api.groq.com/openai/v1', + }, + }) + + expect(result.routeId).toBeNull() + expect(result.descriptor).toBeNull() + expect(result.catalogEntry).toBeNull() + expect(result.openaiShimConfig.removeBodyFields).toBeUndefined() + expect(result.openaiShimConfig.thinkingRequestFormat).toBeUndefined() + }) +}) + describe('resolveOpenAIShimRuntimeContext - segment-boundary heuristic', () => { describe('DeepSeek models', () => { it('should NOT infer preserveReasoningContent for custom aliases (false-positive case)', () => { diff --git a/src/integrations/runtimeMetadata.ts b/src/integrations/runtimeMetadata.ts index 8d5e8c6b2b..fd63d916b9 100644 --- a/src/integrations/runtimeMetadata.ts +++ b/src/integrations/runtimeMetadata.ts @@ -240,11 +240,12 @@ export function resolveOpenAIShimRuntimeContext(options?: { }) const baseUrlRouteId = resolveRouteIdFromBaseUrl(options?.baseUrl) const routeId = - baseUrlRouteId && - (options?.preferBaseUrlRoute || - !activeRouteId || activeRouteId === 'anthropic' || activeRouteId === 'openai') + options?.preferBaseUrlRoute && options.baseUrl !== undefined ? baseUrlRouteId - : activeRouteId + : baseUrlRouteId && + (!activeRouteId || activeRouteId === 'anthropic' || activeRouteId === 'openai') + ? baseUrlRouteId + : activeRouteId const descriptor = routeId && routeId !== 'anthropic' ? getRouteDescriptor(routeId) diff --git a/src/services/api/openaiShim.ts b/src/services/api/openaiShim.ts index e9b047925a..d62aeda4c5 100644 --- a/src/services/api/openaiShim.ts +++ b/src/services/api/openaiShim.ts @@ -38,7 +38,10 @@ import { } from '../../utils/codexCredentials.js' import { logForDebugging } from '../../utils/debug.js' import { isBareMode, isEnvTruthy } from '../../utils/envUtils.js' -import { resolveOpenAIShimReasoningRequestPlan } from '../../utils/effort.js' +import { + resolveModelReasoningControl, + resolveOpenAIShimReasoningRequestPlan, +} from '../../utils/effort.js' import { resolveGeminiCredential } from '../../utils/geminiAuth.js' import { hydrateGeminiAccessTokenFromSecureStorage } from '../../utils/geminiCredentials.js' import { hydrateGithubModelsTokenFromSecureStorage } from '../../utils/githubModelsCredentials.js' @@ -2434,12 +2437,20 @@ class OpenAIShimMessages { ), }) + const reasoningControl = resolveModelReasoningControl(request.resolvedModel, { + routeId: runtimeShimContext.routeId, + useRuntimeFallback: false, + openaiShimConfig: shimConfig, + }) const reasoningRequestPlan = resolveOpenAIShimReasoningRequestPlan({ model: request.resolvedModel, requestedEffort: request.reasoning?.effort, requestThinkingType: (params.thinking as { type?: string } | undefined)?.type, defaultThinkingType: request.thinking?.type, thinkingRequestFormat: shimConfig.thinkingRequestFormat, + routeId: runtimeShimContext.routeId, + useRuntimeFallback: false, + reasoningControl, }) const body: Record = { diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index 1569d1e217..0279e60e11 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -652,11 +652,12 @@ test('OpenAI shim reasoning request plan centralizes DeepSeek and Z.AI serializa }) }) -test('explicit non-generic metadata wire formats stay non-controllable until planner support exists', async () => { +test('explicit compat metadata wire formats are controllable and feed the request planner', async () => { const { modelSupportsEffort, modelSupportsWireEffort, resolveModelReasoningControl, + resolveOpenAIShimReasoningRequestPlan, } = await importFreshEffortModule({ provider: 'openai', supportsCodexReasoningEffort: false, @@ -675,12 +676,24 @@ test('explicit non-generic metadata wire formats stay non-controllable until pla ], }) - expect(resolveModelReasoningControl('custom-deepseek-model')).toMatchObject({ + const reasoningControl = resolveModelReasoningControl('custom-deepseek-model') + expect(reasoningControl).toMatchObject({ supportsReasoning: true, - controllable: false, + controllable: true, source: 'metadata', wireFormat: 'deepseek_compatible', }) - expect(modelSupportsEffort('custom-deepseek-model')).toBe(false) - expect(modelSupportsWireEffort('custom-deepseek-model')).toBe(false) + expect(modelSupportsEffort('custom-deepseek-model')).toBe(true) + expect(modelSupportsWireEffort('custom-deepseek-model')).toBe(true) + expect(resolveOpenAIShimReasoningRequestPlan({ + model: 'custom-deepseek-model', + requestedEffort: 'xhigh', + requestThinkingType: 'enabled', + reasoningControl, + })).toEqual({ + thinkingType: 'enabled', + reasoningEffort: 'max', + wireFormat: 'deepseek_compatible', + source: 'metadata', + }) }) diff --git a/src/utils/effort.ts b/src/utils/effort.ts index a8958322ee..4b1e43f8b5 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -8,6 +8,7 @@ import { get3PModelCapabilityOverride } from './model/modelSupportOverrides.js' import { getAntModelOverrideConfig, resolveAntModel } from './model/antModels.js' import { supportsCodexReasoningEffort } from '../services/api/providerConfig.js' import { + ensureIntegrationsLoaded, getCatalogEntriesForRoute, getModel, resolveActiveRouteIdFromEnv, @@ -123,10 +124,9 @@ function normalizeReasoningDefaultLevel( function metadataWireFormatSupportsEffort( wireFormat: ReasoningWireFormat | undefined, ): boolean { - // Explicit metadata is controllable only when the planner consumes that exact - // wire format directly. DeepSeek/Z.AI formats are currently enabled through - // temporary compatibility rules, not catalog metadata. - return wireFormat === 'reasoning_effort' + return wireFormat === 'reasoning_effort' || + wireFormat === 'deepseek_compatible' || + wireFormat === 'zai_compatible' } function normalizedBaseModel(model: string | undefined): string { @@ -300,6 +300,7 @@ function resolveCatalogReasoningMetadata( return undefined } + ensureIntegrationsLoaded() const normalizedModel = model.trim().split('?', 1)[0]!.trim().toLowerCase() const entries = context?.catalogEntries ?? getCatalogEntriesForRoute(routeId) const entry = entries.find(catalogEntry => @@ -535,11 +536,29 @@ export function resolveOpenAIShimReasoningRequestPlan(options: { requestThinkingType?: string defaultThinkingType?: string thinkingRequestFormat?: OpenAIShimThinkingRequestFormat + routeId?: string | null + useRuntimeFallback?: boolean + reasoningControl?: Pick }): OpenAIShimReasoningRequestPlan { - const wireFormat = resolveCompatibilityWireFormat( - options.model, - options.thinkingRequestFormat, - ) + const metadataWireFormat = options.reasoningControl?.source === 'metadata' + ? options.reasoningControl.wireFormat + : undefined + if (metadataWireFormat && !metadataWireFormatSupportsEffort(metadataWireFormat)) { + return { + wireFormat: metadataWireFormat, + source: 'none', + } + } + + const wireFormat = metadataWireFormat + ? metadataWireFormat + : resolveCompatibilityWireFormat( + options.model, + options.thinkingRequestFormat, + options.routeId, + options.useRuntimeFallback ?? true, + ) + const source = metadataWireFormat ? 'metadata' : 'compat' const requestedThinkingType = normalizeReasoningThinkingType( options.requestThinkingType, ) @@ -556,7 +575,7 @@ export function resolveOpenAIShimReasoningRequestPlan(options: { thinkingType, reasoningEffort, wireFormat, - source: 'compat', + source, } } @@ -566,19 +585,29 @@ export function resolveOpenAIShimReasoningRequestPlan(options: { return { thinkingType: 'disabled', wireFormat, - source: 'compat', + source, } } const shouldEnableThinking = thinkingType === 'enabled' || options.requestedEffort !== undefined - const reasoningEffort = options.requestedEffort && supportsZaiReasoningEffort(options.model) + const metadataZaiSupportsReasoningEffort = + metadataWireFormat === 'zai_compatible' && + (options.reasoningControl?.levels.includes('xhigh') || + options.reasoningControl?.levels.includes('max') || + options.reasoningControl?.levels.includes('medium') || + options.reasoningControl?.levels.includes('low')) + const reasoningEffort = options.requestedEffort && + (metadataZaiSupportsReasoningEffort || ( + metadataWireFormat !== 'zai_compatible' && + supportsZaiReasoningEffort(options.model) + )) ? normalizeZaiReasoningEffort(options.requestedEffort) : undefined return { thinkingType: shouldEnableThinking ? 'enabled' : undefined, reasoningEffort, wireFormat, - source: 'compat', + source, } } From 2fb5ba6ee99484490e53b4c9a68c79f2a100d15b Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 16:24:50 -0700 Subject: [PATCH 7/9] Fix Z.AI metadata high effort serialization Include high in the Z.AI-compatible metadata gate so high-only reasoning metadata emits reasoning_effort instead of silently dropping the user-selected effort. Add regression coverage for a high-only zai_compatible catalog entry flowing through the OpenAI shim request planner. --- src/utils/effort.codex.test.ts | 31 +++++++++++++++++++++++++++++++ src/utils/effort.ts | 3 ++- 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index 0279e60e11..24c6ecb6dd 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -673,6 +673,16 @@ test('explicit compat metadata wire formats are controllable and feed the reques wireFormat: 'deepseek_compatible', }, }, + { + id: 'custom-zai-high-only', + apiName: 'custom-zai-high-only', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['high'], + wireFormat: 'zai_compatible', + }, + }, ], }) @@ -696,4 +706,25 @@ test('explicit compat metadata wire formats are controllable and feed the reques wireFormat: 'deepseek_compatible', source: 'metadata', }) + + const zaiReasoningControl = resolveModelReasoningControl('custom-zai-high-only') + expect(zaiReasoningControl).toMatchObject({ + supportsReasoning: true, + controllable: true, + source: 'metadata', + wireFormat: 'zai_compatible', + levels: ['high'], + }) + expect(modelSupportsEffort('custom-zai-high-only')).toBe(true) + expect(modelSupportsWireEffort('custom-zai-high-only')).toBe(true) + expect(resolveOpenAIShimReasoningRequestPlan({ + model: 'custom-zai-high-only', + requestedEffort: 'high', + reasoningControl: zaiReasoningControl, + })).toEqual({ + thinkingType: 'enabled', + reasoningEffort: 'high', + wireFormat: 'zai_compatible', + source: 'metadata', + }) }) diff --git a/src/utils/effort.ts b/src/utils/effort.ts index 4b1e43f8b5..e3fea16bff 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -592,7 +592,8 @@ export function resolveOpenAIShimReasoningRequestPlan(options: { const shouldEnableThinking = thinkingType === 'enabled' || options.requestedEffort !== undefined const metadataZaiSupportsReasoningEffort = metadataWireFormat === 'zai_compatible' && - (options.reasoningControl?.levels.includes('xhigh') || + (options.reasoningControl?.levels.includes('high') || + options.reasoningControl?.levels.includes('xhigh') || options.reasoningControl?.levels.includes('max') || options.reasoningControl?.levels.includes('medium') || options.reasoningControl?.levels.includes('low')) From 5c6d121f67d42c9d48ed170212528d78544d23de Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 16:55:14 -0700 Subject: [PATCH 8/9] Fix provider override effort fallback Allow unrecognized providerOverride OpenAI-compatible routes to fall back to legacy effort support instead of dropping user-selected effort. Constrain compat metadata levels to wire-faithful high/xhigh values and clarify reserved reasoning wire formats in docs and descriptors. --- docs/integrations/reasoning-effort.md | 6 ++- src/integrations/descriptors.ts | 5 +++ src/services/api/client.test.ts | 53 +++++++++++++++++++++++++++ src/utils/effort.codex.test.ts | 40 ++++++++++++++++++++ src/utils/effort.ts | 28 +++++++++++--- 5 files changed, 124 insertions(+), 8 deletions(-) diff --git a/docs/integrations/reasoning-effort.md b/docs/integrations/reasoning-effort.md index a8429a9521..ef833b06ea 100644 --- a/docs/integrations/reasoning-effort.md +++ b/docs/integrations/reasoning-effort.md @@ -16,8 +16,6 @@ reasoning: { defaultLevel?: 'low' | 'medium' | 'high' | 'xhigh' | 'max' wireFormat?: | 'reasoning_effort' - | 'reasoning_object' - | 'thinking_type' | 'deepseek_compatible' | 'zai_compatible' | 'none' @@ -46,6 +44,10 @@ Prefer catalog-entry metadata when a gateway route differs from the canonical mo Use `mode: 'always-on'` with `wireFormat: 'none'` for models that emit reasoning but do not have a verified control parameter on that route. +Currently wired metadata formats are `reasoning_effort`, `deepseek_compatible`, and `zai_compatible`. The descriptor type also reserves `reasoning_object` and `thinking_type`, but those formats are not request-plumbed yet and should not be used to enable `/effort`. + +For `deepseek_compatible` and `zai_compatible`, metadata levels must be limited to `high` and/or `xhigh`. These serializers emit provider `high` for `high` and provider `max` for `xhigh`; they cannot faithfully represent `low`, `medium`, or standard `max` as distinct UI levels. + ## Adding Support Before adding `reasoning` metadata for a model: diff --git a/src/integrations/descriptors.ts b/src/integrations/descriptors.ts index 91fe43be18..c441612a0b 100644 --- a/src/integrations/descriptors.ts +++ b/src/integrations/descriptors.ts @@ -57,6 +57,11 @@ export interface CapabilityFlags { export type ReasoningControlMode = 'levels' | 'toggle' | 'always-on' export type ReasoningEffortLevel = 'low' | 'medium' | 'high' | 'xhigh' | 'max' +/** + * reasoning_effort, deepseek_compatible, and zai_compatible are wired into + * request serialization today. Other values are reserved until their serializer + * paths are implemented. + */ export type ReasoningWireFormat = | 'reasoning_effort' | 'reasoning_object' diff --git a/src/services/api/client.test.ts b/src/services/api/client.test.ts index e6b9d37178..d5e11f543a 100644 --- a/src/services/api/client.test.ts +++ b/src/services/api/client.test.ts @@ -1385,6 +1385,59 @@ test('providerOverride OpenAI gpt effort does not fall back to ambient provider' expect(requestBody?.reasoning_effort).toBe('xhigh') }) +test('providerOverride custom OpenAI-compatible gpt effort uses legacy support', async () => { + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + + return new Response( + JSON.stringify({ + id: 'chatcmpl-provider-override-custom-openai', + model: 'gpt-5.4', + choices: [ + { + message: { + role: 'assistant', + content: 'ok', + }, + finish_reason: 'stop', + }, + ], + usage: { + prompt_tokens: 8, + completion_tokens: 3, + total_tokens: 11, + }, + }), + { + headers: { + 'Content-Type': 'application/json', + }, + }, + ) + }) as FetchType + + const client = (await getAnthropicClient({ + maxRetries: 0, + effortValue: 'high', + providerOverride: { + model: 'gpt-5.4', + baseURL: 'https://custom-openai-compatible.example.test/v1', + apiKey: 'provider-test-key', + }, + })) as unknown as ShimClient + + await client.beta.messages.create({ + model: 'unused', + system: 'test system', + messages: [{ role: 'user', content: 'hello' }], + max_tokens: 64, + stream: false, + }) + + expect(requestBody?.reasoning_effort).toBe('high') +}) test('providerOverride Groq DeepSeek does not receive stripped effort override', async () => { let requestBody: Record | undefined diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index 24c6ecb6dd..33055bb2bb 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -683,6 +683,26 @@ test('explicit compat metadata wire formats are controllable and feed the reques wireFormat: 'zai_compatible', }, }, + { + id: 'custom-zai-low-only', + apiName: 'custom-zai-low-only', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['low'], + wireFormat: 'zai_compatible', + }, + }, + { + id: 'custom-deepseek-low-only', + apiName: 'custom-deepseek-low-only', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['low'], + wireFormat: 'deepseek_compatible', + }, + }, ], }) @@ -727,4 +747,24 @@ test('explicit compat metadata wire formats are controllable and feed the reques wireFormat: 'zai_compatible', source: 'metadata', }) + + expect(resolveModelReasoningControl('custom-zai-low-only')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'metadata', + wireFormat: 'zai_compatible', + levels: [], + }) + expect(modelSupportsEffort('custom-zai-low-only')).toBe(false) + expect(modelSupportsWireEffort('custom-zai-low-only')).toBe(false) + + expect(resolveModelReasoningControl('custom-deepseek-low-only')).toMatchObject({ + supportsReasoning: true, + controllable: false, + source: 'metadata', + wireFormat: 'deepseek_compatible', + levels: [], + }) + expect(modelSupportsEffort('custom-deepseek-low-only')).toBe(false) + expect(modelSupportsWireEffort('custom-deepseek-low-only')).toBe(false) }) diff --git a/src/utils/effort.ts b/src/utils/effort.ts index e3fea16bff..9786abfecd 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -80,6 +80,7 @@ export type ReasoningControlContext = OpenAIShimReasoningSupportContext & { const DEFAULT_REASONING_LEVELS: EffortLevel[] = ['low', 'medium', 'high'] const OPENAI_SHIM_COMPAT_LEVELS: EffortLevel[] = ['low', 'medium', 'high', 'xhigh'] +const OPENAI_SHIM_METADATA_COMPAT_LEVELS: EffortLevel[] = ['high', 'xhigh'] function getReasoningApiProvider( context?: ReasoningControlContext, @@ -111,6 +112,17 @@ function normalizeReasoningLevels( return normalized.length > 0 ? normalized : [...DEFAULT_REASONING_LEVELS] } +function normalizeMetadataReasoningLevels( + wireFormat: ReasoningWireFormat | undefined, + levels: ReasoningControlMetadata['levels'] | undefined, +): EffortLevel[] { + const normalized = normalizeReasoningLevels(levels) + if (wireFormat === 'deepseek_compatible' || wireFormat === 'zai_compatible') { + return normalized.filter(level => OPENAI_SHIM_METADATA_COMPAT_LEVELS.includes(level)) + } + return normalized +} + function normalizeReasoningDefaultLevel( level: ReasoningControlMetadata['defaultLevel'] | undefined, levels: EffortLevel[], @@ -346,10 +358,10 @@ function resolveMetadataReasoningControl( } } + const wireFormat = reasoning.wireFormat const levels = reasoning.mode === 'levels' - ? normalizeReasoningLevels(reasoning.levels) + ? normalizeMetadataReasoningLevels(wireFormat, reasoning.levels) : [] - const wireFormat = reasoning.wireFormat const controllable = Boolean( capabilities?.supportsReasoning !== false && metadataWireFormatSupportsEffort(wireFormat) && @@ -512,6 +524,13 @@ export function modelSupportsShimReasoningEffort( } if (context?.useRuntimeFallback === false) { + if ( + context.routeId == null && + thinkingRequestFormat === undefined && + !removeBodyFields?.includes('reasoning_effort') + ) { + return resolveLegacyReasoningControl(model, context).controllable + } return false } @@ -593,10 +612,7 @@ export function resolveOpenAIShimReasoningRequestPlan(options: { const metadataZaiSupportsReasoningEffort = metadataWireFormat === 'zai_compatible' && (options.reasoningControl?.levels.includes('high') || - options.reasoningControl?.levels.includes('xhigh') || - options.reasoningControl?.levels.includes('max') || - options.reasoningControl?.levels.includes('medium') || - options.reasoningControl?.levels.includes('low')) + options.reasoningControl?.levels.includes('xhigh')) const reasoningEffort = options.requestedEffort && (metadataZaiSupportsReasoningEffort || ( metadataWireFormat !== 'zai_compatible' && From 78d98f784c4ef1c9314bd3b6f80847c7b6244105 Mon Sep 17 00:00:00 2001 From: jatmn Date: Wed, 24 Jun 2026 17:37:18 -0700 Subject: [PATCH 9/9] Clamp provider override effort by route metadata Resolve providerOverride effort against the override model and route context before converting it for the OpenAI shim, so stale persisted effort values respect per-model metadata levels. Add regression coverage for high-only providerOverride metadata and explicit max filtering in compat metadata levels. --- src/services/api/client.test.ts | 87 +++++++++++++++++++++++++++++++++ src/services/api/client.ts | 29 ++++++++--- src/utils/effort.codex.test.ts | 18 +++++++ 3 files changed, 128 insertions(+), 6 deletions(-) diff --git a/src/services/api/client.test.ts b/src/services/api/client.test.ts index d5e11f543a..ac4a1bce22 100644 --- a/src/services/api/client.test.ts +++ b/src/services/api/client.test.ts @@ -1,5 +1,10 @@ import { afterEach, beforeEach, expect, test } from 'bun:test' import { acquireSharedMutationLock, releaseSharedMutationLock } from '../../test/sharedMutationLock.js' +import { + _clearRegistryForTesting, + ensureIntegrationsLoaded, + registerGateway, +} from '../../integrations/index.js' import { getAnthropicClient } from './client.js' type FetchType = typeof globalThis.fetch @@ -1438,6 +1443,88 @@ test('providerOverride custom OpenAI-compatible gpt effort uses legacy support', expect(requestBody?.reasoning_effort).toBe('high') }) +test('providerOverride clamps stale effort against metadata levels', async () => { + let requestBody: Record | undefined + + globalThis.fetch = (async (_input, init) => { + requestBody = JSON.parse(String(init?.body)) + + return new Response( + JSON.stringify({ + id: 'chatcmpl-provider-override-metadata-clamp', + model: 'metadata-high-only-model', + choices: [ + { + message: { + role: 'assistant', + content: 'ok', + }, + finish_reason: 'stop', + }, + ], + usage: { + prompt_tokens: 8, + completion_tokens: 3, + total_tokens: 11, + }, + }), + { + headers: { + 'Content-Type': 'application/json', + }, + }, + ) + }) as FetchType + + _clearRegistryForTesting() + try { + registerGateway({ + id: 'metadata-effort-test', + label: 'Metadata Effort Test', + defaultBaseUrl: 'https://metadata-effort.example.test/v1', + setup: { requiresAuth: true, authMode: 'api-key' }, + transportConfig: { kind: 'openai-compatible' }, + catalog: { + source: 'static', + models: [ + { + id: 'metadata-high-only-model', + apiName: 'metadata-high-only-model', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['high'], + wireFormat: 'reasoning_effort', + }, + }, + ], + }, + }) + + const client = (await getAnthropicClient({ + maxRetries: 0, + effortValue: 'low', + providerOverride: { + model: 'metadata-high-only-model', + baseURL: 'https://metadata-effort.example.test/v1', + apiKey: 'provider-test-key', + }, + })) as unknown as ShimClient + + await client.beta.messages.create({ + model: 'unused', + system: 'test system', + messages: [{ role: 'user', content: 'hello' }], + max_tokens: 64, + stream: false, + }) + } finally { + _clearRegistryForTesting() + ensureIntegrationsLoaded() + } + + expect(requestBody?.reasoning_effort).toBe('high') +}) test('providerOverride Groq DeepSeek does not receive stripped effort override', async () => { let requestBody: Record | undefined diff --git a/src/services/api/client.ts b/src/services/api/client.ts index 6a49104840..cad19c6782 100644 --- a/src/services/api/client.ts +++ b/src/services/api/client.ts @@ -12,6 +12,7 @@ import { import { convertEffortValueToLevel, type EffortValue, + resolveAppliedEffort, modelSupportsShimReasoningEffort, modelSupportsWireEffort, standardEffortToOpenAI, @@ -342,22 +343,38 @@ export async function getAnthropicClient({ }) : undefined const providerOverrideShimConfig = providerOverrideRuntimeContext?.openaiShimConfig + const providerOverrideEffortContext = providerOverrideRuntimeContext + ? { + routeId: providerOverrideRuntimeContext.routeId, + useRuntimeFallback: false, + openaiShimConfig: providerOverrideShimConfig, + apiProvider: providerOverrideRuntimeContext.routeId === 'openai' + ? 'openai' as const + : providerOverrideRuntimeContext.routeId === 'codex' + ? 'codex' as const + : undefined, + } + : undefined const supportsShimReasoningEffort = effortModel ? providerOverrideShimConfig ? modelSupportsShimReasoningEffort( effortModel, providerOverrideShimConfig.thinkingRequestFormat, providerOverrideShimConfig.removeBodyFields, - { - routeId: providerOverrideRuntimeContext?.routeId, - useRuntimeFallback: false, - }, + providerOverrideEffortContext, ) : modelSupportsWireEffort(effortModel) : false + const appliedProviderOverrideEffort = effortModel && effortValue !== undefined + ? resolveAppliedEffort( + effortModel, + effortValue, + providerOverrideEffortContext, + ) + : undefined const shimReasoningEffort: OpenAIEffortLevel | undefined = - effortValue !== undefined && supportsShimReasoningEffort - ? standardEffortToOpenAI(convertEffortValueToLevel(effortValue)) + appliedProviderOverrideEffort !== undefined && supportsShimReasoningEffort + ? standardEffortToOpenAI(convertEffortValueToLevel(appliedProviderOverrideEffort)) : undefined const containerId = process.env.CLAUDE_CODE_CONTAINER_ID const remoteSessionId = process.env.CLAUDE_CODE_REMOTE_SESSION_ID diff --git a/src/utils/effort.codex.test.ts b/src/utils/effort.codex.test.ts index 33055bb2bb..e15758345a 100644 --- a/src/utils/effort.codex.test.ts +++ b/src/utils/effort.codex.test.ts @@ -673,6 +673,16 @@ test('explicit compat metadata wire formats are controllable and feed the reques wireFormat: 'deepseek_compatible', }, }, + { + id: 'custom-deepseek-with-max', + apiName: 'custom-deepseek-with-max', + capabilities: { supportsReasoning: true }, + reasoning: { + mode: 'levels', + levels: ['high', 'max', 'xhigh'], + wireFormat: 'deepseek_compatible', + }, + }, { id: 'custom-zai-high-only', apiName: 'custom-zai-high-only', @@ -727,6 +737,14 @@ test('explicit compat metadata wire formats are controllable and feed the reques source: 'metadata', }) + expect(resolveModelReasoningControl('custom-deepseek-with-max')).toMatchObject({ + supportsReasoning: true, + controllable: true, + source: 'metadata', + wireFormat: 'deepseek_compatible', + levels: ['high', 'xhigh'], + }) + const zaiReasoningControl = resolveModelReasoningControl('custom-zai-high-only') expect(zaiReasoningControl).toMatchObject({ supportsReasoning: true,