Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions docs/advanced-setup.md
Original file line number Diff line number Diff line change
Expand Up @@ -494,6 +494,12 @@ Model env vars are provider-scoped: first-party Anthropic sessions read
`GEMINI_MODEL`, and Mistral reads `MISTRAL_MODEL`. For manual Bedrock, Vertex,
or Foundry launches, select the model with `--model`.

An OpenAI-compatible provider profile's maximum context length applies to every
model in its configured list and the supported saved `/model` selection restored
for that profile. Query options such as `?reasoning=high` or `?thinking=disabled`
remain in the selected model, but context-limit keys use the model name before
`?`. For example, `gpt-5.4?reasoning=high` uses the `gpt-5.4` context limit.

### Per-model limit overrides (`settings.json`)

When a custom OpenAI-compatible provider does not expose context metadata from
Expand Down
176 changes: 144 additions & 32 deletions src/utils/providerProfiles.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2139,24 +2139,37 @@ describe('applyProviderProfileToProcessEnv', () => {
expect(getFreshAPIProvider()).not.toBe('xai')
})

test('openai-compatible profile applies maxContextLength env override', async () => {
test('openai-compatible profile applies maxContextLength to every configured model', async () => {
const { applyProviderProfileToProcessEnv } =
await importFreshProviderProfileModules()
const { resolveModelRuntimeLimits } = await import(
'../integrations/runtimeMetadata.js'
)

applyProviderProfileToProcessEnv(
buildProfile({
provider: 'custom',
baseUrl: 'http://localhost:4000/v1',
model: 'gpt-4o',
model: 'local-large, local-small; local-reasoning',
maxContextLength: 200_000,
}),
)

expect(process.env.OPENAI_BASE_URL).toBe('http://localhost:4000/v1')
expect(process.env.OPENAI_MODEL).toBe('gpt-4o')
expect(process.env.OPENAI_MODEL).toBe('local-large')
expect(process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS).toBe(
JSON.stringify({ 'gpt-4o': 200_000 }),
JSON.stringify({
'local-large': 200_000,
'local-small': 200_000,
'local-reasoning': 200_000,
}),
)
expect(
resolveModelRuntimeLimits({
model: 'local-small',
processEnv: process.env,
}).contextWindow,
).toBe(200_000)
})

test('openai-compatible profile switch clears previous same-model context override', async () => {
Expand Down Expand Up @@ -2977,34 +2990,87 @@ describe('applyActiveProviderProfileFromConfig', () => {
expect(process.env.OPENAI_MODEL).toBe('gpt-4o')
})

test('uses saved valid Hicap /model choice when rehydrating active profile', async () => {
const {
_setSavedModelOverrideForTesting,
applyActiveProviderProfileFromConfig,
getProviderProfiles,
} = await importFreshProviderProfileModules()
_setSavedModelOverrideForTesting('gpt-5.4')
const activeProfile = buildProfile({
id: 'saved_hicap',
provider: 'hicap',
baseUrl: 'https://api.hicap.ai/v1',
model: 'glm-5.2',
})
test.each(['', '?reasoning=high', '?thinking=disabled&reasoning=high'])(
'preserves context limits for configured and saved models across profile rehydration (%j)',
async query => {
const {
_setSavedModelOverrideForTesting,
applyActiveProviderProfileFromConfig,
getProviderProfiles,
} = await importFreshProviderProfileModules()
const { resolveModelRuntimeLimits } = await import(
'../integrations/runtimeMetadata.js'
)
const configuredModel = `gpt-5.2${query}`
const savedModel = `gpt-5.4${query}`
const modelList = `glm-5.2; ${configuredModel}`
_setSavedModelOverrideForTesting(savedModel)
const activeProfile = buildProfile({
id: 'saved_hicap',
provider: 'hicap',
baseUrl: 'https://api.hicap.ai/v1',
model: modelList,
maxContextLength: 200_000,
})
const config = {
providerProfiles: [activeProfile],
activeProviderProfileId: activeProfile.id,
} as any
const expectedContextWindows = JSON.stringify({
'glm-5.2': 200_000,
'gpt-5.2': 200_000,
'gpt-5.4': 200_000,
})

const applied = applyActiveProviderProfileFromConfig({
providerProfiles: [activeProfile],
activeProviderProfileId: activeProfile.id,
} as any)
const applied = applyActiveProviderProfileFromConfig(config)

expect(applied?.id).toBe(activeProfile.id)
expect(process.env.OPENAI_BASE_URL).toBe('https://api.hicap.ai/v1')
const expectProfileContextLimits = () => {
expect(process.env.OPENAI_MODEL).toBe(savedModel)
for (const model of [
'glm-5.2',
'gpt-5.2',
'gpt-5.4',
configuredModel,
savedModel,
]) {
expect(
resolveModelRuntimeLimits({ model, processEnv: process.env }).contextWindow,
).toBe(200_000)
}
expect(process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS).toBe(
expectedContextWindows,
)
const saved = getProviderProfiles(config).find(
(profile: ProviderProfile) => profile.id === activeProfile.id,
)
expect(saved?.model).toBe(modelList)
}
expectProfileContextLimits()

expect(applied?.id).toBe(activeProfile.id)
expect(process.env.OPENAI_BASE_URL).toBe('https://api.hicap.ai/v1')
expect(process.env.OPENAI_MODEL).toBe('gpt-5.4')
const saved = getProviderProfiles({
providerProfiles: [activeProfile],
activeProviderProfileId: activeProfile.id,
} as any).find((profile: ProviderProfile) => profile.id === activeProfile.id)
expect(saved?.model).toBe('glm-5.2')
})
expect(applyActiveProviderProfileFromConfig(config)?.id).toBe(activeProfile.id)
expectProfileContextLimits()

// Alignment must repair the old configured-only map, even though the
// effective model and all other profile-managed environment values match.
process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = JSON.stringify({
'glm-5.2': 200_000,
'gpt-5.2': 200_000,
})
applyActiveProviderProfileFromConfig(config)
expectProfileContextLimits()

// Also repair a complete map written with the old raw query-bearing keys.
process.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = JSON.stringify({
'glm-5.2': 200_000,
[configuredModel]: 200_000,
[savedModel]: 200_000,
})
applyActiveProviderProfileFromConfig(config)
expectProfileContextLimits()
},
)

test('uses saved Codex /model choice when rehydrating the Codex OAuth profile', async () => {
// Regression: the Codex OAuth profile is created with a single
Expand Down Expand Up @@ -3697,9 +3763,10 @@ describe('setActiveProviderProfile', () => {
name: 'DeepSeek',
provider: 'openai',
baseUrl: 'https://api.deepseek.com/v1',
model: 'deepseek-v4-flash, deepseek-v4-pro, deepseek-chat',
model: 'deepseek-v4-flash?thinking=disabled, deepseek-v4-pro?reasoning=high, deepseek-chat',
apiKey: 'sk-deepseek-live',
apiFormat: 'responses',
maxContextLength: 200_000,
})

saveMockGlobalConfig(current => ({
Expand All @@ -3719,8 +3786,13 @@ describe('setActiveProviderProfile', () => {
expect(persisted.profile).toBe('openai')
expect(persisted.env).toEqual({
OPENAI_BASE_URL: 'https://api.deepseek.com/v1',
OPENAI_MODEL: 'deepseek-v4-flash',
OPENAI_MODEL: 'deepseek-v4-flash?thinking=disabled',
OPENAI_API_KEY: 'sk-deepseek-live',
CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: JSON.stringify({
'deepseek-v4-flash': 200_000,
'deepseek-v4-pro': 200_000,
'deepseek-chat': 200_000,
}),
})
} finally {
process.chdir(originalCwd)
Expand All @@ -3729,6 +3801,46 @@ describe('setActiveProviderProfile', () => {
}
})

test('persists context window for every model in a keyless openai-compatible profile', async () => {
const configDir = mkdtempSync(join(tmpdir(), 'openclaude-provider-config-'))

try {
const { setActiveProviderProfile } =
await importFreshProviderProfileModules()
const openaiProfile = buildProfile({
id: 'keyless_multi_model_prof',
provider: 'custom',
baseUrl: 'http://localhost:4000/v1',
model: 'model-a?reasoning=high; Model-B[1m]?thinking=disabled, model-c',
maxContextLength: 64_000,
})

saveMockGlobalConfig(current => ({
...current,
providerProfiles: [openaiProfile],
}))

const result = setActiveProviderProfile('keyless_multi_model_prof', {
configDir,
})
const persisted = JSON.parse(
readFileSync(join(configDir, '.openclaude-profile.json'), 'utf8'),
)

expect(result?.id).toBe('keyless_multi_model_prof')
expect(persisted.env.OPENAI_MODEL).toBe('model-a?reasoning=high')
expect(persisted.env.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS).toBe(
JSON.stringify({
'model-a': 64_000,
'Model-B[1m]': 64_000,
'model-c': 64_000,
}),
)
} finally {
rmSync(configDir, { recursive: true, force: true })
}
})

test('persists descriptor-backed direct vendors using a legacy-compatible openai startup profile', async () => {
const tempDir = mkdtempSync(join(tmpdir(), 'openclaude-provider-'))
const configDir = mkdtempSync(join(tmpdir(), 'openclaude-provider-config-'))
Expand Down
59 changes: 50 additions & 9 deletions src/utils/providerProfiles.ts
Original file line number Diff line number Diff line change
Expand Up @@ -735,6 +735,34 @@ function sameOptionalEnvValue(
return trimOrUndefined(left) === trimOrUndefined(right)
}

function serializeProfileContextWindows(
modelField: string,
maxContextLength: number,
activeModel?: string,
): string {
const models = parseModelList(modelField)
const configuredModels =
models.length > 0 ? models : [getPrimaryModel(modelField)]
// A saved /model choice can be valid for the provider catalog without
// being listed in profile.model. Include the model that will actually run
// so it does not fall back to a different context limit.
const selectedModel = trimOrUndefined(activeModel)
const contextModels = selectedModel
? [...new Set([...configuredModels, selectedModel])]
: configuredModels

return JSON.stringify(
Object.fromEntries(
// Match resolveModelRuntimeLimits' query-stripped lookup key without
// changing case, [1m] tags, or the raw model selection and its options.
contextModels.map(model => [
model.split('?', 1)[0]?.trim() || model,
maxContextLength,
]),
),
)
}

function isProcessEnvAlignedWithProfile(
processEnv: NodeJS.ProcessEnv,
profile: ProviderProfile,
Expand Down Expand Up @@ -850,9 +878,11 @@ function isProcessEnvAlignedWithProfile(
}

const expectedContextWindows = profile.maxContextLength
? JSON.stringify({
[primaryModel]: profile.maxContextLength,
})
? serializeProfileContextWindows(
profile.model,
profile.maxContextLength,
primaryModel,
)
: undefined
const isAimlapiRoute =
profile.provider === 'aimlapi' ||
Expand Down Expand Up @@ -1281,9 +1311,12 @@ export function applyProviderProfileToProcessEnv(
openAIProfileEnv.NVIDIA_NIM = '1'
}
if (profile.maxContextLength) {
openAIProfileEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS = JSON.stringify({
[primaryModel]: profile.maxContextLength,
})
openAIProfileEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS =
serializeProfileContextWindows(
profile.model,
profile.maxContextLength,
primaryModel,
)
}

profileEnv = openAIProfileEnv
Expand Down Expand Up @@ -1599,6 +1632,13 @@ function buildOpenAICompatibleStartupEnv(
processEnv: {},
})
if (strictEnv) {
if (activeProfile.maxContextLength) {
strictEnv.CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS =
serializeProfileContextWindows(
activeProfile.model,
activeProfile.maxContextLength,
)
}
if (isAimlapiProfile) {
strictEnv.AIMLAPI_API_KEY = activeProfile.apiKey
strictEnv.CLAUDE_CODE_PROVIDER_ROUTE_ID = 'aimlapi'
Expand Down Expand Up @@ -1657,9 +1697,10 @@ function buildOpenAICompatibleStartupEnv(
...(activeProfile.authHeaderValue ? { OPENAI_AUTH_HEADER_VALUE: activeProfile.authHeaderValue } : {}),
...(activeProfile.maxContextLength
? {
CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: JSON.stringify({
[getPrimaryModel(activeProfile.model)]: activeProfile.maxContextLength,
}),
CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: serializeProfileContextWindows(
activeProfile.model,
activeProfile.maxContextLength,
),
}
: {}),
}
Expand Down
Loading