diff --git a/docs/advanced-setup.md b/docs/advanced-setup.md index f0cc2c06e2..4428a63e47 100644 --- a/docs/advanced-setup.md +++ b/docs/advanced-setup.md @@ -522,12 +522,19 @@ addition to the `CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS` / key is `localhost:4000:my-model`, not `localhost:my-model`. Either field may be omitted to override only one limit. - **Precedence** — from highest to lowest: an **exact** env-var override → the - built-in catalog value → the discovery-cache value → a **prefix** env-var - override → `modelLimits` → the descriptor default. (The built-in catalog is - checked before the discovery cache.) So env-var overrides always win over - `modelLimits`, and `modelLimits` mainly fills in models that have no built-in - metadata (a known catalog model keeps its catalog limit unless you set an - *exact* env override for it). + built-in catalog value → a **prefix** env-var override → `modelLimits` → the + discovery-cache value → the descriptor default. So env-var overrides and + `modelLimits` both win over discovery. A known catalog model keeps its catalog + limit unless you set an *exact* env override for it. +- **When a gateway advertises the wrong window** — some OpenAI-compatible + gateways report a flat `context_length` (often `128000`) for every model they + proxy. OpenClaude keeps that value, because an advertised window can just as + easily be a real per-deployment cap, and budgeting above the endpoint's true + limit turns early auto-compact into a mid-session API failure. If you know the + real window, use an **exact** `CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS` entry for a + catalogued model. For a custom or discovered model, use `modelLimits` or an + exact env entry. Use `/set-context-window ` for the current session + only. ### Exact-model pricing overrides (`settings.json`) diff --git a/src/commands/set-context-window/set-context-window.ts b/src/commands/set-context-window/set-context-window.ts index b432355f69..1cfcfa0af9 100644 --- a/src/commands/set-context-window/set-context-window.ts +++ b/src/commands/set-context-window/set-context-window.ts @@ -14,8 +14,15 @@ Examples: /set-context-window gpt-4o 200000 /set-context-window status -The override takes precedence over integration catalog and provider -profile defaults. Use /clear-context-window to remove it.` +The override takes precedence over discovery, catalog, and provider defaults +for this session only. Use /clear-context-window to remove it. + +For a permanent override (survives restart), use modelLimits in settings.json +for custom/discovered models, for example: + { "modelLimits": { "my-model": { "contextWindow": 1000000 } } } +For a catalogued model, use an exact CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS entry; +catalog values take precedence over modelLimits and prefix env entries. +See docs/advanced-setup.md (Per-model limit overrides).` export const call: LocalCommandCall = async (args, context) => { const trimmed = args.trim() diff --git a/src/integrations/runtimeMetadata.modelLimits.test.ts b/src/integrations/runtimeMetadata.modelLimits.test.ts index 284a8180ba..60b7944674 100644 --- a/src/integrations/runtimeMetadata.modelLimits.test.ts +++ b/src/integrations/runtimeMetadata.modelLimits.test.ts @@ -162,3 +162,76 @@ test('resolveModelRuntimeLimits lets a broad env-prefix override win over an exa expect(limits.contextWindow).toBe(111_111) expect(limits.maxOutputTokens).toBe(4_096) }) + +test('resolveModelRuntimeLimits lets settings modelLimits beat discovery cache', async () => { + const { clearDiscoveryCache, setCachedModels } = await import( + `./discoveryCache.js?ts=${Date.now()}` + ) + const { getDiscoveryCacheKey } = await import( + `./discoveryService.js?ts=${Date.now()}` + ) + // The discovery cache path comes from getClaudeConfigHomeDir(), which ignores + // CLAUDE_CONFIG_DIR by design. Without this override the fixture below would + // be written into the caller's real ~/.openclaude. + const { + getClaudeConfigHomeDirOverrideForTesting, + setClaudeConfigHomeDirForTesting, + } = await import('../utils/envUtils.js') + const { mkdtempSync, rmSync } = await import('node:fs') + const { tmpdir } = await import('node:os') + const { join } = await import('node:path') + + const previousOverride = getClaudeConfigHomeDirOverrideForTesting() + const tempDir = mkdtempSync(join(tmpdir(), 'openclaude-model-limits-')) + setClaudeConfigHomeDirForTesting(tempDir) + try { + const baseUrl = 'http://localhost:20128/v1' + await setCachedModels( + getDiscoveryCacheKey('custom', { baseUrl }), + { + models: [ + { + id: 'my-codex-combo', + apiName: 'my-codex-combo', + label: 'my-codex-combo', + contextWindow: 128_000, + maxOutputTokens: 8_192, + }, + ], + }, + ) + + const { resolveModelRuntimeLimits } = await importFresh() + const resolveWithDiscovery = () => + resolveModelRuntimeLimits({ + model: 'my-codex-combo', + processEnv: { + CLAUDE_CODE_USE_OPENAI: '1', + OPENAI_BASE_URL: baseUrl, + }, + }) + + // Establish that the isolated cache is observable first, so the override + // assertion below proves precedence rather than passing on a missing fixture. + mockSettings = {} + const discovered = resolveWithDiscovery() + expect(discovered.contextWindow).toBe(128_000) + expect(discovered.maxOutputTokens).toBe(8_192) + + mockSettings = { + modelLimits: { + 'my-codex-combo': { contextWindow: 1_000_000, maxOutputTokens: 32_768 }, + }, + } + const overridden = resolveWithDiscovery() + + expect(overridden.contextWindow).toBe(1_000_000) + expect(overridden.maxOutputTokens).toBe(32_768) + } finally { + // Clears the module-level sync snapshot as well, so the fixture cannot leak + // into a later suite once the temp dir is removed. + await clearDiscoveryCache() + setClaudeConfigHomeDirForTesting(previousOverride) + rmSync(tempDir, { recursive: true, force: true }) + } +}) diff --git a/src/integrations/runtimeMetadata.test.ts b/src/integrations/runtimeMetadata.test.ts index 99425f9e9b..713a3bde2e 100644 --- a/src/integrations/runtimeMetadata.test.ts +++ b/src/integrations/runtimeMetadata.test.ts @@ -11,19 +11,20 @@ import { resolveModelRuntimeLimits, resolveOpenAIShimRuntimeContext, } from '../integrations/runtimeMetadata' -import { setCachedModels } from './discoveryCache' +import { clearDiscoveryCache, setCachedModels } from './discoveryCache' +import { + getClaudeConfigHomeDirOverrideForTesting, + setClaudeConfigHomeDirForTesting, +} from '../utils/envUtils.js' import { getDiscoveryCacheKey, getRouteDiscoveryHeaders, } from './discoveryService' import { resolveActiveRouteIdFromEnv } from './routeMetadata.js' -import { setClaudeConfigHomeDirForTesting } from '../utils/envUtils.js' import glmBrand from './brands/glm.js' import glmModels from './models/glm.js' import zaiVendor from './vendors/zai.js' -const originalConfigDir = process.env.CLAUDE_CONFIG_DIR - describe('Z.AI GLM-5.3 descriptor contract', () => { it('wires the verified GLM-5.3-Flash descriptor and direct catalog contract', () => { const model = glmModels.find(candidate => candidate.id === 'glm-5.3-flash') @@ -114,25 +115,27 @@ describe('Z.AI GLM-5.3 descriptor contract', () => { }) }) +// The discovery cache resolves its path through getClaudeConfigHomeDir(), which +// deliberately ignores CLAUDE_CONFIG_DIR (OpenClaude config is independent of +// Claude Code config). Use the explicit test override so these fixtures land in +// the temp dir instead of the developer's real ~/.openclaude. async function withTempConfigDir(fn: () => Promise): Promise { await acquireSharedMutationLock('integrations/runtimeMetadata.test.ts') + const previousOverride = getClaudeConfigHomeDirOverrideForTesting() let tempDir: string | null = null try { tempDir = mkdtempSync(join(tmpdir(), 'openclaude-runtime-metadata-test-')) setClaudeConfigHomeDirForTesting(tempDir) - process.env.CLAUDE_CONFIG_DIR = tempDir return await fn() } finally { try { - if (originalConfigDir === undefined) { - delete process.env.CLAUDE_CONFIG_DIR - } else { - process.env.CLAUDE_CONFIG_DIR = originalConfigDir - } + // Drops the module-level sync snapshot so a later suite cannot read this + // suite's fixture back out of memory after the temp dir is gone. + await clearDiscoveryCache() + setClaudeConfigHomeDirForTesting(previousOverride) if (tempDir) { rmSync(tempDir, { recursive: true, force: true }) } - setClaudeConfigHomeDirForTesting(undefined) } finally { releaseSharedMutationLock() } @@ -420,6 +423,123 @@ describe('resolveModelRuntimeLimits', () => { ).toBe(2_000_000) }) }) + + it.each([128_000, 200_000])( + 'keeps a gateway-advertised %i window for a globally larger known model', + async advertised => { + // User-reported path: OmniRoute + GPT-5.6 Sol ("gpt sol"), whose gateway + // advertised a smaller window than the known descriptor. An advertised + // `context_length` is indistinguishable from a real per-deployment cap, so + // discovery stays authoritative — budgeting past the endpoint's real limit + // would trade an early compact for a mid-session API failure. 128k is + // covered explicitly because it is the value gateways most often report as + // a flat default. + await withTempConfigDir(async () => { + const baseUrl = 'http://localhost:20128/v1' + await setCachedModels( + getDiscoveryCacheKey('custom', { + baseUrl, + }), + { + models: [ + { + id: 'gpt-5.6-sol', + apiName: 'gpt-5.6-sol', + label: 'gpt-5.6-sol', + contextWindow: advertised, + }, + ], + }, + ) + + expect( + resolveModelRuntimeLimits({ + model: 'gpt-5.6-sol', + processEnv: { + CLAUDE_CODE_USE_OPENAI: '1', + OPENAI_BASE_URL: baseUrl, + }, + }).contextWindow, + ).toBe(advertised) + }) + }, + ) + + it('lets an exact env override beat a wrong discovery-cache context window', async () => { + await withTempConfigDir(async () => { + const baseUrl = 'http://localhost:20128/v1' + await setCachedModels( + getDiscoveryCacheKey('custom', { + baseUrl, + }), + { + models: [ + { + id: 'my-codex-combo', + apiName: 'my-codex-combo', + label: 'my-codex-combo', + contextWindow: 128_000, + }, + ], + }, + ) + + expect( + resolveModelRuntimeLimits({ + model: 'my-codex-combo', + processEnv: { + CLAUDE_CODE_USE_OPENAI: '1', + OPENAI_BASE_URL: baseUrl, + CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: JSON.stringify({ + 'my-codex-combo': 1_000_000, + }), + }, + }).contextWindow, + ).toBe(1_000_000) + }) + }) + + it('lets env prefix overrides beat discovery for both context and max-output limits', async () => { + // CodeRabbit on #2082: exact-key and settings.modelLimits vs discovery are + // covered, but not CLAUDE_CODE_OPENAI_* prefix keys. A prefix pin is the + // documented way to cover a family of gateway model ids without listing + // each one, and must sit above the discovery cache for both limits. + await withTempConfigDir(async () => { + const baseUrl = 'http://localhost:20128/v1' + await setCachedModels( + getDiscoveryCacheKey('custom', { + baseUrl, + }), + { + models: [ + { + id: 'my-codex-combo-v2', + apiName: 'my-codex-combo-v2', + label: 'my-codex-combo-v2', + contextWindow: 128_000, + maxOutputTokens: 8_192, + }, + ], + }, + ) + + expect( + resolveModelRuntimeLimits({ + model: 'my-codex-combo-v2', + processEnv: { + CLAUDE_CODE_USE_OPENAI: '1', + OPENAI_BASE_URL: baseUrl, + CLAUDE_CODE_OPENAI_CONTEXT_WINDOWS: JSON.stringify({ + 'my-codex-combo': 1_000_000, + }), + CLAUDE_CODE_OPENAI_MAX_OUTPUT_TOKENS: JSON.stringify({ + 'my-codex-combo': 32_768, + }), + }, + }), + ).toEqual({ contextWindow: 1_000_000, maxOutputTokens: 32_768 }) + }) + }) }) describe('LLMTR runtime attribution', () => { diff --git a/src/integrations/runtimeMetadata.ts b/src/integrations/runtimeMetadata.ts index 50fa438599..bf4d79f453 100644 --- a/src/integrations/runtimeMetadata.ts +++ b/src/integrations/runtimeMetadata.ts @@ -531,28 +531,39 @@ export function resolveModelRuntimeLimits(options: { runtimeEnv, ) - // Precedence: an exact env override wins outright; then the built-in - // catalog / discovery-cache value (a `:cloud` variant must take its known - // catalog limit rather than inherit a broad base-model env *prefix*); then a - // broad env *prefix* override; then the settings.json `modelLimits` override; - // then the descriptor default. The key fix for the env/settings drift is - // keeping `settings` strictly below `prefix` so a broad env-prefix override is + // Precedence (high → low): + // 1. exact env override + // 2. built-in route catalog (so `:cloud` variants keep their catalog cap over + // a broad base-model env *prefix*) + // 3. env *prefix* override + // 4. settings.json `modelLimits` (explicit user pin) + // 5. discovery cache + // 6. model descriptor default + // Discovery stays authoritative over the descriptor: a gateway's advertised + // `context_length` is the endpoint's real cap (it may legitimately be a + // smaller deployment/tenant limit for a globally larger model), and the + // OpenAI-compatible response carries no signal that would let us tell a + // synthetic gateway default apart from a real cap. Users whose gateway + // advertises a wrong window pin it via an exact env override, `modelLimits` + // for an uncatalogued model, or `/set-context-window`; each applicable + // override sits above discovery here. + // Keep `settings` strictly below `prefix` so a broad env-prefix override is // never silently overtaken by a settings entry — matching the scalar // getOpenAIContextWindow, where env (exact or prefix) beats settings. return { contextWindow: externalContextWindow.exact ?? catalogEntry?.contextWindow ?? - cachedCatalogEntry?.contextWindow ?? externalContextWindow.prefix ?? externalContextWindow.settings ?? + cachedCatalogEntry?.contextWindow ?? modelDescriptor?.contextWindow, maxOutputTokens: externalMaxOutputTokens.exact ?? catalogEntry?.maxOutputTokens ?? - cachedCatalogEntry?.maxOutputTokens ?? externalMaxOutputTokens.prefix ?? externalMaxOutputTokens.settings ?? + cachedCatalogEntry?.maxOutputTokens ?? modelDescriptor?.maxOutputTokens, } } diff --git a/src/utils/model/openaiContextWindows.ts b/src/utils/model/openaiContextWindows.ts index e9bea25018..16cc354042 100644 --- a/src/utils/model/openaiContextWindows.ts +++ b/src/utils/model/openaiContextWindows.ts @@ -9,8 +9,8 @@ * This module only produces the *override candidates* (exact/prefix env-var * matches and the settings `modelLimits` match); it does not decide the overall * precedence. The authoritative runtime chain — exact env override, then the - * catalog/discovery cache, then the prefix env override, then settings - * `modelLimits`, then the descriptor default — is applied by + * built-in catalog, then the prefix env override, then settings `modelLimits`, + * then the discovery cache, then the descriptor default — is applied by * resolveModelRuntimeLimits in integrations/runtimeMetadata.ts. Keep precedence * changes there, not duplicated here. */ @@ -27,7 +27,7 @@ export type OpenAILimitOverrideMatches = { // settings.json `modelLimits` match (exact or prefix). Just a candidate here; // its position in the overall precedence is decided by resolveModelRuntimeLimits // (integrations/runtimeMetadata.ts), which applies settings after the exact and - // prefix env overrides and the catalog/discovery cache. + // prefix env overrides and catalog, but before the discovery cache. settings?: number // Prefix env-var override match. prefix?: number