diff --git a/docs/advanced-setup.md b/docs/advanced-setup.md index 82062038c8..6d073f3960 100644 --- a/docs/advanced-setup.md +++ b/docs/advanced-setup.md @@ -494,6 +494,12 @@ Model env vars are provider-scoped: first-party Anthropic sessions read `GEMINI_MODEL`, and Mistral reads `MISTRAL_MODEL`. For manual Bedrock, Vertex, or Foundry launches, select the model with `--model`. +An OpenAI-compatible provider profile's maximum context length applies to every +model in its configured list and the supported saved `/model` selection restored +for that profile. Query options such as `?reasoning=high` or `?thinking=disabled` +remain in the selected model, but context-limit keys use the model name before +`?`. For example, `gpt-5.4?reasoning=high` uses the `gpt-5.4` context limit. + ### Per-model limit overrides (`settings.json`) When a custom OpenAI-compatible provider does not expose context metadata from diff --git a/src/utils/providerProfiles.test.ts b/src/utils/providerProfiles.test.ts index 06b77adf1c..d4848f9204 100644 --- a/src/utils/providerProfiles.test.ts +++ b/src/utils/providerProfiles.test.ts @@ -2139,24 +2139,37 @@ describe('applyProviderProfileToProcessEnv', () => { expect(getFreshAPIProvider()).not.toBe('xai') }) - test('openai-compatible profile applies maxContextLength env override', async () => { + test('openai-compatible profile applies maxContextLength to every configured model', async () => { const { applyProviderProfileToProcessEnv } = await importFreshProviderProfileModules() + const { resolveModelRuntimeLimits } = await import( + '../integrations/runtimeMetadata.js' + ) applyProviderProfileToProcessEnv( buildProfile({ provider: 'custom', baseUrl: 'http://localhost:4000/v1', - model: 'gpt-4o', + model: 'local-large, local-small; local-reasoning', maxContextLength: 200_000, }), ) expect(process.env.OPENAI_BASE_URL).toBe('http://localhost:4000/v1') - expect(process.env.OPENAI_MODEL).toBe('gpt-4o') + expect(process.env.OPENAI_MODEL).toBe('local-large') expect(process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS).toBe( - JSON.stringify({ 'gpt-4o': 200_000 }), + JSON.stringify({ + 'local-large': 200_000, + 'local-small': 200_000, + 'local-reasoning': 200_000, + }), ) + expect( + resolveModelRuntimeLimits({ + model: 'local-small', + processEnv: process.env, + }).contextWindow, + ).toBe(200_000) }) test('openai-compatible profile switch clears previous same-model context override', async () => { @@ -2977,34 +2990,87 @@ describe('applyActiveProviderProfileFromConfig', () => { expect(process.env.OPENAI_MODEL).toBe('gpt-4o') }) - test('uses saved valid Hicap /model choice when rehydrating active profile', async () => { - const { - _setSavedModelOverrideForTesting, - applyActiveProviderProfileFromConfig, - getProviderProfiles, - } = await importFreshProviderProfileModules() - _setSavedModelOverrideForTesting('gpt-5.4') - const activeProfile = buildProfile({ - id: 'saved_hicap', - provider: 'hicap', - baseUrl: 'https://api.hicap.ai/v1', - model: 'glm-5.2', - }) + test.each(['', '?reasoning=high', '?thinking=disabled&reasoning=high'])( + 'preserves context limits for configured and saved models across profile rehydration (%j)', + async query => { + const { + _setSavedModelOverrideForTesting, + applyActiveProviderProfileFromConfig, + getProviderProfiles, + } = await importFreshProviderProfileModules() + const { resolveModelRuntimeLimits } = await import( + '../integrations/runtimeMetadata.js' + ) + const configuredModel = `gpt-5.2${query}` + const savedModel = `gpt-5.4${query}` + const modelList = `glm-5.2; ${configuredModel}` + _setSavedModelOverrideForTesting(savedModel) + const activeProfile = buildProfile({ + id: 'saved_hicap', + provider: 'hicap', + baseUrl: 'https://api.hicap.ai/v1', + model: modelList, + maxContextLength: 200_000, + }) + const config = { + providerProfiles: [activeProfile], + activeProviderProfileId: activeProfile.id, + } as any + const expectedContextWindows = JSON.stringify({ + 'glm-5.2': 200_000, + 'gpt-5.2': 200_000, + 'gpt-5.4': 200_000, + }) - const applied = applyActiveProviderProfileFromConfig({ - providerProfiles: [activeProfile], - activeProviderProfileId: activeProfile.id, - } as any) + const applied = applyActiveProviderProfileFromConfig(config) + + expect(applied?.id).toBe(activeProfile.id) + expect(process.env.OPENAI_BASE_URL).toBe('https://api.hicap.ai/v1') + const expectProfileContextLimits = () => { + expect(process.env.OPENAI_MODEL).toBe(savedModel) + for (const model of [ + 'glm-5.2', + 'gpt-5.2', + 'gpt-5.4', + configuredModel, + savedModel, + ]) { + expect( + resolveModelRuntimeLimits({ model, processEnv: process.env }).contextWindow, + ).toBe(200_000) + } + expect(process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS).toBe( + expectedContextWindows, + ) + const saved = getProviderProfiles(config).find( + (profile: ProviderProfile) => profile.id === activeProfile.id, + ) + expect(saved?.model).toBe(modelList) + } + expectProfileContextLimits() - expect(applied?.id).toBe(activeProfile.id) - expect(process.env.OPENAI_BASE_URL).toBe('https://api.hicap.ai/v1') - expect(process.env.OPENAI_MODEL).toBe('gpt-5.4') - const saved = getProviderProfiles({ - providerProfiles: [activeProfile], - activeProviderProfileId: activeProfile.id, - } as any).find((profile: ProviderProfile) => profile.id === activeProfile.id) - expect(saved?.model).toBe('glm-5.2') - }) + expect(applyActiveProviderProfileFromConfig(config)?.id).toBe(activeProfile.id) + expectProfileContextLimits() + + // Alignment must repair the old configured-only map, even though the + // effective model and all other profile-managed environment values match. + process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = JSON.stringify({ + 'glm-5.2': 200_000, + 'gpt-5.2': 200_000, + }) + applyActiveProviderProfileFromConfig(config) + expectProfileContextLimits() + + // Also repair a complete map written with the old raw query-bearing keys. + process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = JSON.stringify({ + 'glm-5.2': 200_000, + [configuredModel]: 200_000, + [savedModel]: 200_000, + }) + applyActiveProviderProfileFromConfig(config) + expectProfileContextLimits() + }, + ) test('uses saved Codex /model choice when rehydrating the Codex OAuth profile', async () => { // Regression: the Codex OAuth profile is created with a single @@ -3697,9 +3763,10 @@ describe('setActiveProviderProfile', () => { name: 'DeepSeek', provider: 'openai', baseUrl: 'https://api.deepseek.com/v1', - model: 'deepseek-v4-flash, deepseek-v4-pro, deepseek-chat', + model: 'deepseek-v4-flash?thinking=disabled, deepseek-v4-pro?reasoning=high, deepseek-chat', apiKey: 'sk-deepseek-live', apiFormat: 'responses', + maxContextLength: 200_000, }) saveMockGlobalConfig(current => ({ @@ -3719,8 +3786,13 @@ describe('setActiveProviderProfile', () => { expect(persisted.profile).toBe('openai') expect(persisted.env).toEqual({ OPENAI_BASE_URL: 'https://api.deepseek.com/v1', - OPENAI_MODEL: 'deepseek-v4-flash', + OPENAI_MODEL: 'deepseek-v4-flash?thinking=disabled', OPENAI_API_KEY: 'sk-deepseek-live', + CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: JSON.stringify({ + 'deepseek-v4-flash': 200_000, + 'deepseek-v4-pro': 200_000, + 'deepseek-chat': 200_000, + }), }) } finally { process.chdir(originalCwd) @@ -3729,6 +3801,46 @@ describe('setActiveProviderProfile', () => { } }) + test('persists context window for every model in a keyless openai-compatible profile', async () => { + const configDir = mkdtempSync(join(tmpdir(), 'openclaude-provider-config-')) + + try { + const { setActiveProviderProfile } = + await importFreshProviderProfileModules() + const openaiProfile = buildProfile({ + id: 'keyless_multi_model_prof', + provider: 'custom', + baseUrl: 'http://localhost:4000/v1', + model: 'model-a?reasoning=high; Model-B[1m]?thinking=disabled, model-c', + maxContextLength: 64_000, + }) + + saveMockGlobalConfig(current => ({ + ...current, + providerProfiles: [openaiProfile], + })) + + const result = setActiveProviderProfile('keyless_multi_model_prof', { + configDir, + }) + const persisted = JSON.parse( + readFileSync(join(configDir, '.openclaude-profile.json'), 'utf8'), + ) + + expect(result?.id).toBe('keyless_multi_model_prof') + expect(persisted.env.OPENAI_MODEL).toBe('model-a?reasoning=high') + expect(persisted.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS).toBe( + JSON.stringify({ + 'model-a': 64_000, + 'Model-B[1m]': 64_000, + 'model-c': 64_000, + }), + ) + } finally { + rmSync(configDir, { recursive: true, force: true }) + } + }) + test('persists descriptor-backed direct vendors using a legacy-compatible openai startup profile', async () => { const tempDir = mkdtempSync(join(tmpdir(), 'openclaude-provider-')) const configDir = mkdtempSync(join(tmpdir(), 'openclaude-provider-config-')) diff --git a/src/utils/providerProfiles.ts b/src/utils/providerProfiles.ts index 113402c42c..7a8a92ac77 100644 --- a/src/utils/providerProfiles.ts +++ b/src/utils/providerProfiles.ts @@ -735,6 +735,34 @@ function sameOptionalEnvValue( return trimOrUndefined(left) === trimOrUndefined(right) } +function serializeProfileContextWindows( + modelField: string, + maxContextLength: number, + activeModel?: string, +): string { + const models = parseModelList(modelField) + const configuredModels = + models.length > 0 ? models : [getPrimaryModel(modelField)] + // A saved /model choice can be valid for the provider catalog without + // being listed in profile.model. Include the model that will actually run + // so it does not fall back to a different context limit. + const selectedModel = trimOrUndefined(activeModel) + const contextModels = selectedModel + ? [...new Set([...configuredModels, selectedModel])] + : configuredModels + + return JSON.stringify( + Object.fromEntries( + // Match resolveModelRuntimeLimits' query-stripped lookup key without + // changing case, [1m] tags, or the raw model selection and its options. + contextModels.map(model => [ + model.split('?', 1)[0]?.trim() || model, + maxContextLength, + ]), + ), + ) +} + function isProcessEnvAlignedWithProfile( processEnv: NodeJS.ProcessEnv, profile: ProviderProfile, @@ -850,9 +878,11 @@ function isProcessEnvAlignedWithProfile( } const expectedContextWindows = profile.maxContextLength - ? JSON.stringify({ - [primaryModel]: profile.maxContextLength, - }) + ? serializeProfileContextWindows( + profile.model, + profile.maxContextLength, + primaryModel, + ) : undefined const isAimlapiRoute = profile.provider === 'aimlapi' || @@ -1281,9 +1311,12 @@ export function applyProviderProfileToProcessEnv( openAIProfileEnv.NVIDIA_NIM = '1' } if (profile.maxContextLength) { - openAIProfileEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = JSON.stringify({ - [primaryModel]: profile.maxContextLength, - }) + openAIProfileEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = + serializeProfileContextWindows( + profile.model, + profile.maxContextLength, + primaryModel, + ) } profileEnv = openAIProfileEnv @@ -1599,6 +1632,13 @@ function buildOpenAICompatibleStartupEnv( processEnv: {}, }) if (strictEnv) { + if (activeProfile.maxContextLength) { + strictEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = + serializeProfileContextWindows( + activeProfile.model, + activeProfile.maxContextLength, + ) + } if (isAimlapiProfile) { strictEnv.AIMLAPI_API_KEY = activeProfile.apiKey strictEnv.CLAUDE_CODE_PROVIDER_ROUTE_ID = 'aimlapi' @@ -1657,9 +1697,10 @@ function buildOpenAICompatibleStartupEnv( ...(activeProfile.authHeaderValue ? { OPENAI_AUTH_HEADER_VALUE: activeProfile.authHeaderValue } : {}), ...(activeProfile.maxContextLength ? { - CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: JSON.stringify({ - [getPrimaryModel(activeProfile.model)]: activeProfile.maxContextLength, - }), + CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: serializeProfileContextWindows( + activeProfile.model, + activeProfile.maxContextLength, + ), } : {}), }