diff --git a/changelog.d/features/15171-gpt-61-sol-catalog-pricing.md b/changelog.d/features/15171-gpt-61-sol-catalog-pricing.md new file mode 100644 index 00000000000..1998156a2cb --- /dev/null +++ b/changelog.d/features/15171-gpt-61-sol-catalog-pricing.md @@ -0,0 +1 @@ +- **feat(providers):** Add GPT-6.1 Sol to the OpenAI and Codex catalogs with provider-specific limits, Standard pricing, Codex reasoning/Fast wiring, and a synchronized Codex 0.159.2 client identity ([#15171](https://github.com/diegosouzapw/OmniRoute/pull/15171)) — thanks @xiaoyaner0201 diff --git a/open-sse/config/providers/registry/openai/index.ts b/open-sse/config/providers/registry/openai/index.ts index 4aa30a17427..70d7d58bd85 100644 --- a/open-sse/config/providers/registry/openai/index.ts +++ b/open-sse/config/providers/registry/openai/index.ts @@ -12,6 +12,14 @@ export const openaiProvider: RegistryEntry = { authHeader: "bearer", defaultContextLength: 128000, models: [ + // https://developers.openai.com/api/docs/models/gpt-6.1-sol + // API: 1.05M context, low..max (no none/minimal); tools require Responses. + { + id: "gpt-6.1-sol", + name: "GPT-6.1 Sol", + ...GPT_5_6_API_CAPABILITIES, + unsupportedParams: ["temperature", "top_p", "top_logprobs", "logprobs"], + }, // Astra shares the public GPT-5.6 limits; tool calling requires Responses. // https://developers.openai.com/api/docs/guides/latest-model { @@ -33,7 +41,6 @@ export const openaiProvider: RegistryEntry = { { id: "gpt-5.6-sol", name: "GPT-5.6 Sol", ...GPT_5_6_API_CAPABILITIES }, { id: "gpt-5.6-terra", name: "GPT-5.6 Terra", ...GPT_5_6_API_CAPABILITIES }, { id: "gpt-5.6-luna", name: "GPT-5.6 Luna", ...GPT_5_6_API_CAPABILITIES }, - { id: "gpt-6.1-sol", name: "GPT 6.1 Sol", ...GPT_5_6_API_CAPABILITIES }, { id: "gpt-5.5", name: "GPT-5.5", contextLength: 1050000 }, // #5842: *-pro reasoning models are responses-only upstream — /v1/chat/completions // 404s ("only supported in v1/responses"). targetFormat routes them natively. diff --git a/open-sse/executors/codex/reasoningSuffix.ts b/open-sse/executors/codex/reasoningSuffix.ts index f2f6162a35d..ebb2c9f1e2c 100644 --- a/open-sse/executors/codex/reasoningSuffix.ts +++ b/open-sse/executors/codex/reasoningSuffix.ts @@ -9,6 +9,7 @@ export const CODEX_EFFORT_ORDER = [ ] as const; export type CodexEffortLevel = (typeof CODEX_EFFORT_ORDER)[number]; export const CODEX_MAX_ALIAS_MODELS = new Set([ + "gpt-6.1-sol", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", @@ -18,6 +19,7 @@ export const CODEX_MAX_ALIAS_MODELS = new Set([ "gpt-6-luna", ]); export const CODEX_ULTRA_ALIAS_MODELS = new Set([ + "gpt-6.1-sol", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-6-astra", diff --git a/src/lib/providers/codexFastTier.ts b/src/lib/providers/codexFastTier.ts index 3715b81ffab..45d9447bf1d 100644 --- a/src/lib/providers/codexFastTier.ts +++ b/src/lib/providers/codexFastTier.ts @@ -14,6 +14,7 @@ export type CodexFastTierValue = CodexServiceTier; export type CodexGlobalServiceMode = "none" | CodexServiceTier; export const CODEX_FAST_TIER_DEFAULT_SUPPORTED_MODELS: readonly string[] = [ + "gpt-6.1-sol", "gpt-6-astra", "gpt-6-sol", "gpt-6-luna", diff --git a/src/shared/constants/modelSpecs.ts b/src/shared/constants/modelSpecs.ts index 443dd435269..706da33a28b 100644 --- a/src/shared/constants/modelSpecs.ts +++ b/src/shared/constants/modelSpecs.ts @@ -128,6 +128,11 @@ const GEMINI_36_FLASH_MODEL_SPEC = { } satisfies ModelSpec; export const MODEL_SPECS: Record = { + // Public API limits; Codex's smaller window lives in its provider registry. + "gpt-6.1-sol": { + ...GPT_5_6_MODEL_SPEC, + aliases: ["openai/gpt-6.1-sol"], + }, // Public model limits; the Codex registry supplies its smaller OAuth window. // https://developers.openai.com/api/docs/models/gpt-6-astra "gpt-6-astra": { @@ -146,7 +151,6 @@ export const MODEL_SPECS: Record = { ...GPT_5_6_MODEL_SPEC, aliases: ["openai/gpt-6-luna"], }, - "gpt-6.1-sol": GPT_5_6_MODEL_SPEC, "gpt-5.6": { ...GPT_5_6_MODEL_SPEC, aliases: ["openai/gpt-5.6"], diff --git a/tests/unit/codex-fast-tier.test.ts b/tests/unit/codex-fast-tier.test.ts index f686eeca057..bf65b821765 100644 --- a/tests/unit/codex-fast-tier.test.ts +++ b/tests/unit/codex-fast-tier.test.ts @@ -46,6 +46,7 @@ test("Codex global service mode distinguishes no setting from explicit tiers", ( enabled: true, tier: "default", supportedModels: [ + "gpt-6.1-sol", "gpt-6-astra", "gpt-6-sol", "gpt-6-luna", diff --git a/tests/unit/codex-gpt61-sol-bare-id.test.ts b/tests/unit/codex-gpt61-sol-bare-id.test.ts new file mode 100644 index 00000000000..6d86e0d145a --- /dev/null +++ b/tests/unit/codex-gpt61-sol-bare-id.test.ts @@ -0,0 +1,111 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// GPT-6.1 Sol is a Codex-native model like gpt-6-astra and the gpt-5.6 tiers. +// Its ids must be in CODEX_NATIVE_UNPREFIXED_MODELS, otherwise with both a Codex +// and an OpenAI connection active the bare `gpt-6.1-sol` id is ambiguous (or goes +// to OpenAI) and /v1/models never lists the bare Codex rows. + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-codex-gpt61-bare-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "codex-gpt61-bare-test-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const apiKeysDb = await import("../../src/lib/db/apiKeys.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const { CODEX_NATIVE_UNPREFIXED_MODELS, getModelInfoCore } = + await import("../../open-sse/services/model.ts"); +const { getProviderModels } = await import("../../open-sse/config/providerModels.ts"); +const { getPricingForModel } = await import("../../src/shared/constants/pricing.ts"); +const v1ModelsCatalog = await import("../../src/app/api/v1/models/catalog.ts"); + +const MODEL = "gpt-6.1-sol"; +const EFFORTS = ["ultra", "max", "xhigh", "high", "medium", "low"]; +const EXPECTED_IDS = [MODEL, ...EFFORTS.map((effort) => `${MODEL}-${effort}`)]; + +// The base id and its effort tiers, as registered for the codex provider. +const SOL_IDS = getProviderModels("codex") + .map((model) => model.id) + .filter((id) => id === MODEL || id.startsWith(`${MODEL}-`)); + +async function seedConnection(provider: "codex" | "openai") { + await providersDb.createProviderConnection({ + provider, + authType: provider === "codex" ? "oauth" : "apikey", + name: `${provider}-gpt61-bare`, + email: provider === "codex" ? "codex@example.com" : undefined, + apiKey: provider === "openai" ? "sk-openai-gpt61-bare" : undefined, + accessToken: provider === "codex" ? "codex-gpt61-bare-access" : undefined, + isActive: true, + testStatus: "active", + providerSpecificData: provider === "codex" ? { workspaceId: "ws-gpt61-bare" } : {}, + }); +} + +test.beforeEach(() => { + core.resetDbInstance(); + apiKeysDb.resetApiKeyState(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +}); + +test.after(() => { + core.resetDbInstance(); + apiKeysDb.resetApiKeyState(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("the Codex catalog registers exactly the base GPT-6.1 Sol id and its six effort tiers", () => { + assert.deepEqual([...SOL_IDS].sort(), [...EXPECTED_IDS].sort()); +}); + +test("every GPT-6.1 Sol id in the Codex catalog is Codex-native when unprefixed", () => { + for (const id of SOL_IDS) { + assert.equal(CODEX_NATIVE_UNPREFIXED_MODELS.has(id), true, id); + } +}); + +test("every GPT-6.1 Sol Codex id resolves a non-zero pricing row", () => { + for (const id of SOL_IDS) { + const price = getPricingForModel("cx", id); + assert.ok(price, `cx/${id}`); + assert.ok(price.input > 0 && price.output > 0, `cx/${id} must not resolve to $0`); + } +}); + +test("bare gpt-6.1-sol routes to Codex when Codex and OpenAI are both active", async () => { + await seedConnection("codex"); + await seedConnection("openai"); + + for (const id of [MODEL, `${MODEL}-ultra`, `${MODEL}-low`]) { + const info = await getModelInfoCore(id, null); + assert.equal(info.provider, "codex", id); + assert.equal(info.model, id); + } +}); + +test("bare gpt-6.1-sol stays on OpenAI when no Codex connection is active", async () => { + await seedConnection("openai"); + + const info = await getModelInfoCore(MODEL, null); + assert.equal(info.provider, "openai"); + assert.equal(info.model, MODEL); +}); + +test("/v1/models lists the bare GPT-6.1 Sol ids under their codex/ rows", async () => { + await seedConnection("codex"); + + v1ModelsCatalog.__resetCatalogBuilderRunsForTest(); + const response = await v1ModelsCatalog.getUnifiedModelsResponse( + new Request("http://localhost/api/v1/models") + ); + const body = (await response.json()) as { data: Array<{ id: string; parent?: string | null }> }; + const parentById = new Map(body.data.map((row) => [row.id, row.parent ?? null])); + + for (const id of SOL_IDS) { + assert.equal(parentById.get(id), `codex/${id}`, id); + } +}); diff --git a/tests/unit/gpt61-sol-catalog.test.ts b/tests/unit/gpt61-sol-catalog.test.ts new file mode 100644 index 00000000000..3455f76eef9 --- /dev/null +++ b/tests/unit/gpt61-sol-catalog.test.ts @@ -0,0 +1,177 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { getModelsByProviderId } from "../../open-sse/config/providerModels.ts"; +import { CodexExecutor } from "../../open-sse/executors/codex.ts"; +import { openaiToOpenAIResponsesRequest } from "../../open-sse/translator/request/openai-responses/toResponses.ts"; +import { getPricingForModel } from "../../src/shared/constants/pricing.ts"; +import { getModelSpec, capMaxOutputTokens } from "../../src/shared/constants/modelSpecs.ts"; +import { + computeCostFromPricing, + getCodexFastCostMultiplier, +} from "../../src/lib/usage/costCalculator.ts"; +import { extendCodexGpt56EffortValues } from "../../src/shared/reasoning/effortStandardization.ts"; +import { isCodexExtendedEffortBaseModel } from "../../src/shared/reasoning/codexExtendedEffort.ts"; +import * as reasoningMetadata from "../../src/lib/vscode/reasoningMetadata.ts"; +import { supportsVscodeServiceTierVariants } from "../../src/lib/vscode/serviceTierVariants.ts"; + +const MODEL = "gpt-6.1-sol"; +const EFFORTS = ["low", "medium", "high", "xhigh", "max", "ultra"]; +const IDS = [MODEL, ...EFFORTS.map((effort) => `${MODEL}-${effort}`)]; + +test.after(async () => { + const { resetDbInstance } = await import("../../src/lib/db/core.ts"); + resetDbInstance(); +}); + +type ResponsesBody = Record & { + reasoning?: { effort?: string; summary?: string }; +}; + +function transform(model: string, body: Record = {}): ResponsesBody { + return new CodexExecutor().transformRequest(model, { model, input: [], ...body }, false, { + requestEndpointPath: "/responses", + }) as ResponsesBody; +} + +test("GPT-6.1 Sol has separate API and Codex catalog limits", () => { + for (const provider of ["codex", "codex-app-server"]) { + const entries = getModelsByProviderId(provider).filter((entry) => entry.id.startsWith(MODEL)); + assert.deepEqual(entries.map((entry) => entry.id).sort(), [...IDS].sort()); + for (const entry of entries) { + assert.equal(entry.contextLength, 872000); + assert.equal(entry.maxInputTokens, 872000); + assert.equal(entry.maxOutputTokens, 128000); + assert.equal(entry.targetFormat, "openai-responses"); + assert.equal(entry.supportsVision, true); + assert.equal(entry.supportsReasoning, true); + assert.equal(entry.toolCalling, true); + } + } + const api = getModelsByProviderId("openai").find((entry) => entry.id === MODEL); + assert.ok(api); + assert.equal(api.contextLength, 1050000); + assert.equal(api.maxOutputTokens, 128000); + assert.equal(api.targetFormat, "openai-responses"); + assert.deepEqual(api.supportedThinkingEfforts, EFFORTS.slice(0, -1)); + assert.equal(getModelSpec(`openai/${MODEL}`)?.contextWindow, 1050000); + assert.equal(capMaxOutputTokens(MODEL, 200000), 128000); +}); + +test("GPT-6.1 Sol aliases and chat translation preserve max on the Codex wire", () => { + for (const effort of EFFORTS) { + const result = transform(`${MODEL}-${effort}`); + assert.equal(result.model, MODEL); + assert.equal(result.reasoning.effort, effort === "ultra" ? "max" : effort); + } + for (const effort of ["max", "ultra"]) { + assert.equal(isCodexExtendedEffortBaseModel(`cx/${MODEL}`, effort as "max" | "ultra"), true); + const result = transform(`${MODEL}(${effort})`, { reasoning: { summary: "detailed" } }); + assert.equal(result.model, MODEL); + assert.equal(result.reasoning.effort, "max"); + assert.equal(result.reasoning.summary, "detailed"); + } + const translated = openaiToOpenAIResponsesRequest( + MODEL, + { + model: MODEL, + messages: [{ role: "user", content: "test" }], + reasoning_effort: "max", + }, + true, + {} + ) as ResponsesBody; + assert.equal(translated.reasoning.effort, "max"); + assert.equal(transform(MODEL, translated).reasoning.effort, "max"); +}); + +test("GPT-6.1 Sol uses the catalog medium default without changing explicit effort or older models", () => { + assert.equal(transform(MODEL).reasoning.effort, "medium"); + assert.equal(transform(MODEL, { reasoning: { effort: "high" } }).reasoning.effort, "high"); + assert.equal(transform("gpt-6-sol").reasoning.effort, "medium"); + for (const provider of ["codex", "cx"]) { + assert.deepEqual(extendCodexGpt56EffortValues(provider, MODEL, ["none", "high"]), EFFORTS); + const model = { + id: `${provider}/${MODEL}`, + owned_by: provider, + capabilities: { reasoning: true }, + }; + assert.deepEqual(reasoningMetadata.getReasoningEffortValues(model), EFFORTS); + assert.equal(reasoningMetadata.getDefaultReasoningEffort(model), "medium"); + assert.equal(reasoningMetadata.getReasoningVariantBaseModelId(`${model.id}-ultra`), model.id); + assert.equal(supportsVscodeServiceTierVariants(model), true); + } + assert.deepEqual(extendCodexGpt56EffortValues("kiro", MODEL, ["high"]), ["high"]); + assert.deepEqual( + extendCodexGpt56EffortValues("openai", MODEL, ["low", "medium", "high", "xhigh", "max"]), + EFFORTS.slice(0, -1) + ); +}); + +test("GPT-6.1 Sol USD pricing uses the current Standard rates and the GPT-6 Fast multiplier", () => { + for (const [provider, ids] of [ + ["openai", [MODEL]], + ["cx", IDS], + ] as const) { + for (const id of ids) { + const price = getPricingForModel(provider, id); + assert.ok(price, `${provider}/${id}`); + assert.equal(price.input, 2); + assert.equal(price.cached, 0.1); + assert.equal(price.output, 10); + assert.equal(price.reasoning, 10); + if (provider === "openai") assert.equal(price.cache_creation, 2.5); + assert.equal( + getCodexFastCostMultiplier(provider, id, "priority"), + provider === "cx" ? 2.5 : 1 + ); + } + } + assert.equal(getCodexFastCostMultiplier("codex", `${MODEL}-ultra`, "fast"), 2.5); + assert.equal(getCodexFastCostMultiplier("codex", MODEL, "default"), 1); + const price = getPricingForModel("cx", MODEL); + assert.ok(price); + // 500 uncached input + 500 cached input + 100 output = $0.00205, Fast (2.5x) = $0.005125. + const tokens = { prompt_tokens: 1000, cached_tokens: 500, completion_tokens: 100 }; + assert.ok( + Math.abs( + computeCostFromPricing(price, tokens, { + provider: "codex", + model: MODEL, + serviceTier: "priority", + }) - 0.005125 + ) < 1e-12 + ); +}); + +test("Responses Lite keeps delegation for GPT-6.1 Sol ultra only", async () => { + for (const effort of ["ultra", "max"]) { + const captured: Record[] = []; + const originalFetch = globalThis.fetch; + globalThis.fetch = async (_url, init) => { + captured.push(JSON.parse(String(init?.body || "{}"))); + return new Response(JSON.stringify({ id: "resp_test", object: "response" }), { + headers: { "Content-Type": "application/json" }, + }); + }; + try { + await new CodexExecutor().execute({ + model: `${MODEL}-${effort}`, + body: { + model: `${MODEL}-${effort}`, + input: [], + _nativeCodexPassthrough: true, + parallel_tool_calls: true, + }, + stream: true, + credentials: { accessToken: "test-token" }, + clientHeaders: { "X-OpenAI-Internal-Codex-Responses-Lite": "true" }, + }); + } finally { + globalThis.fetch = originalFetch; + } + assert.equal(captured.length, 1); + assert.equal(captured[0].model, MODEL); + assert.equal(captured[0].parallel_tool_calls, effort === "ultra"); + } +}); diff --git a/tests/unit/openai-gpt56-catalog.test.ts b/tests/unit/openai-gpt56-catalog.test.ts index 4beb61b97bb..6cd34d7a8de 100644 --- a/tests/unit/openai-gpt56-catalog.test.ts +++ b/tests/unit/openai-gpt56-catalog.test.ts @@ -13,6 +13,7 @@ import { openaiToOpenAIResponsesRequest } from "../../open-sse/translator/reques // gpt-6-sol and gpt-6-luna were added by #15023 (they were missing and fell back to // the 128k default, corrupting combo context windows). const EXPECTED_MODELS = [ + "gpt-6.1-sol", "gpt-6-astra", "gpt-6-sol", "gpt-6-luna",