Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -363,7 +363,7 @@ RUN --mount=type=cache,id=s/92ca8a61-c1ba-421f-a389-d48ac7258c2d-apt-cache,targe
# build, not the floating `@latest`.
RUN --mount=type=cache,id=s/92ca8a61-c1ba-421f-a389-d48ac7258c2d-npm-cache,target=/root/.npm \
npm install -g --no-audit --no-fund \
@openai/codex@0.156.1 \
@openai/codex@0.159.2 \
@anthropic-ai/claude-code@2.1.260 \
droid@0.212.0 \
openclaw@2026.9.1
Expand Down
9 changes: 9 additions & 0 deletions open-sse/config/providers/registry/codex/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,15 @@ export const codexProvider: RegistryEntry = {
},
{ id: "gpt-6-sol-medium", name: "GPT 6 Sol (Medium)", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6-sol-low", name: "GPT 6 Sol (Low)", ...GPT_5_6_CODEX_CAPABILITIES },
// GPT-6.1 Sol: low..ultra effort aliases, same windows as Sol.
// https://developers.openai.com/api/docs/models/gpt-6.1-sol
{ id: "gpt-6.1-sol", name: "GPT 6.1 Sol", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6.1-sol-ultra", name: "GPT 6.1 Sol (Ultra)", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6.1-sol-max", name: "GPT 6.1 Sol (Max)", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6.1-sol-xhigh", name: "GPT 6.1 Sol (xHigh)", ...GPT_5_6_CODEX_CAPABILITIES, timeoutMs: 1200000 },
{ id: "gpt-6.1-sol-high", name: "GPT 6.1 Sol (High)", ...GPT_5_6_CODEX_CAPABILITIES, timeoutMs: 1200000 },
{ id: "gpt-6.1-sol-medium", name: "GPT 6.1 Sol (Medium)", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6.1-sol-low", name: "GPT 6.1 Sol (Low)", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6-luna", name: "GPT 6 Luna", ...GPT_5_6_CODEX_CAPABILITIES },
{ id: "gpt-6-luna-max", name: "GPT 6 Luna (Max)", ...GPT_5_6_CODEX_CAPABILITIES },
{
Expand Down
18 changes: 18 additions & 0 deletions open-sse/config/providers/registry/ghe-copilot/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -124,6 +124,24 @@ export const gheCopilotProvider: RegistryEntry = {
contextLength: 1000000,
maxOutputTokens: 64000,
},
{
id: "gpt-6-astra",
name: "GPT-6 Astra",
supportsReasoning: true,
supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max", "ultra"],
targetFormat: "openai-responses",
contextLength: 1050000,
maxOutputTokens: 128000,
},
{
id: "gpt-6.1-sol",
name: "GPT-6.1 Sol",
supportsReasoning: true,
supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max", "ultra"],
targetFormat: "openai-responses",
contextLength: 1050000,
maxOutputTokens: 128000,
},
{
id: "gpt-5.6-sol",
name: "GPT-5.6 Sol",
Expand Down
18 changes: 18 additions & 0 deletions open-sse/config/providers/registry/github/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -160,6 +160,24 @@ export const githubProvider: RegistryEntry = {
contextLength: 1000000,
maxOutputTokens: 64000,
},
{
id: "gpt-6-astra",
name: "GPT-6 Astra",
supportsReasoning: true,
supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max", "ultra"],
targetFormat: "openai-responses",
contextLength: 1050000,
maxOutputTokens: 128000,
},
{
id: "gpt-6.1-sol",
name: "GPT-6.1 Sol",
supportsReasoning: true,
supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max", "ultra"],
targetFormat: "openai-responses",
contextLength: 1050000,
maxOutputTokens: 128000,
},
{
id: "gpt-5.6-sol",
name: "GPT-5.6 Sol",
Expand Down
2 changes: 2 additions & 0 deletions open-sse/executors/codex/reasoningSuffix.ts
Original file line number Diff line number Diff line change
Expand Up @@ -15,12 +15,14 @@ export const CODEX_MAX_ALIAS_MODELS = new Set([
"gpt-6-astra",
"gpt-6-sol",
"gpt-6-luna",
"gpt-6.1-sol",
]);
export const CODEX_ULTRA_ALIAS_MODELS = new Set([
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-6-astra",
"gpt-6-sol",
"gpt-6.1-sol",
]);

/** Highest effort a max/ultra-tier base model accepts, or null for other models. */
Expand Down
2 changes: 1 addition & 1 deletion open-sse/executors/github.ts
Original file line number Diff line number Diff line change
Expand Up @@ -110,7 +110,7 @@ export class GithubExecutor extends BaseExecutor {
// 9router#1536: but never route Gemini/Claude variants to /responses (they 400) —
// gate the whole decision on supportsResponsesEndpoint().
if (
(targetFormat === "openai-responses" || /codex/i.test(model)) &&
(targetFormat === "openai-responses" || /codex|^gpt-6/i.test(model)) &&
this.supportsResponsesEndpoint(model)
) {
return (
Expand Down
2 changes: 1 addition & 1 deletion open-sse/translator/request/openai-responses/helpers.ts
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ export function imageUrlToText(value: unknown): string {
}

const CODEX_MAX_EFFORT_MODEL_PATTERN =
/^(?:gpt-5\.6-(?:sol|terra|luna)|gpt-6-(?:astra|sol|luna))(?:-(?:none|low|medium|high|xhigh|max|ultra))?$/;
/^(?:gpt-5\.6-(?:sol|terra|luna)|gpt-6(?:\.\d+)?-(?:astra|sol|luna))(?:-(?:none|low|medium|high|xhigh|max|ultra))?$/;
const KIRO_GPT_5_6_MODEL_PATTERN =
/^(?:kiro|kr)\/gpt-5\.6-(?:sol|terra|luna)(?:-(?:none|low|medium|high|xhigh|max))?$/;

Expand Down
1 change: 1 addition & 0 deletions src/lib/providers/codexFastTier.ts
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ export type CodexGlobalServiceMode = "none" | CodexServiceTier;
export const CODEX_FAST_TIER_DEFAULT_SUPPORTED_MODELS: readonly string[] = [
"gpt-6-astra",
"gpt-6-sol",
"gpt-6.1-sol",
"gpt-6-luna",
"gpt-5.6-sol",
"gpt-5.6-terra",
Expand Down
33 changes: 26 additions & 7 deletions src/lib/usage/costCalculator.ts
Original file line number Diff line number Diff line change
Expand Up @@ -103,8 +103,8 @@ export function getCodexFastCostMultiplier(
const compactModelKey = modelKey.replace(/-/g, "");
// Codex GPT-6 Fast is 2.5x Standard (https://developers.openai.com/codex/pricing).
if (
/^gpt-6-(?:astra|sol|luna)$/.test(modelKey) ||
/^gpt6(?:astra|sol|luna)$/.test(compactModelKey)
/^gpt-6(?:\.\d+)?-(?:astra|sol|luna)$/.test(modelKey) ||
/^gpt6(?:\d+)?(?:astra|sol|luna)$/.test(compactModelKey)
) {
return 2.5;
}
Expand All @@ -119,6 +119,19 @@ export function getCodexFastCostMultiplier(
return 1;
}

const GPT61_LONG_CONTEXT_INPUT_TOKENS = 272_000;

function longContextPriceMultiplier(
model: string | null | undefined,
inputTokens: number
): { input: number; cache: number; output: number } {
const flat = { input: 1, cache: 1, output: 1 };
if (inputTokens <= GPT61_LONG_CONTEXT_INPUT_TOKENS) return flat;
const modelKey = stripCodexEffortSuffix(normalizeModelName(String(model || "")).toLowerCase());
if (!/^gpt-6\.\d+-sol$/.test(modelKey)) return flat;
return { input: 2, cache: 2, output: 1.5 };
}

/**
* Calculate cost for a usage entry.
*
Expand Down Expand Up @@ -162,20 +175,26 @@ export function computeCostFromPricing(
// so we must subtract BOTH cache types to avoid pricing cache at the full
// input rate in addition to their dedicated cache_* rates below.
const nonCachedInput = Math.max(0, inputTokens - cachedTokens - cacheCreationTokens);
cost += nonCachedInput * (inputPrice / 1_000_000);
if (cachedTokens > 0) cost += cachedTokens * (cachedPrice / 1_000_000);
// GPT-6.1 bills the whole request at the long-context rate once input
// crosses 272K: 2x input and cache, 1.5x output.
// https://developers.openai.com/api/docs/models/gpt-6.1-sol
const longContext = longContextPriceMultiplier(options.model, inputTokens);
cost += nonCachedInput * ((inputPrice * longContext.input) / 1_000_000);
if (cachedTokens > 0) cost += cachedTokens * ((cachedPrice * longContext.cache) / 1_000_000);

const outputTokens = tokens.output ?? tokens.completion_tokens ?? tokens.output_tokens ?? 0;
cost += outputTokens * (outputPrice / 1_000_000);
cost += outputTokens * ((outputPrice * longContext.output) / 1_000_000);

// completion_tokens is reasoning-inclusive. Reasoning is already billed at
// the output rate above, so a dedicated price contributes only its premium.
const reasoningTokens = tokens.reasoning ?? tokens.reasoning_tokens ?? 0;
if (reasoningTokens > 0 && pricing.reasoning !== undefined && pricing.reasoning !== null) {
cost += reasoningTokens * ((reasoningPrice - outputPrice) / 1_000_000);
cost += reasoningTokens * (Math.max(0, reasoningPrice - outputPrice * longContext.output) / 1_000_000);
}

if (cacheCreationTokens > 0) cost += cacheCreationTokens * (cacheCreationPrice / 1_000_000);
if (cacheCreationTokens > 0) {
cost += cacheCreationTokens * ((cacheCreationPrice * longContext.cache) / 1_000_000);
}

return cost * getCodexFastCostMultiplier(options.provider, options.model, options.serviceTier);
}
Expand Down
2 changes: 1 addition & 1 deletion src/shared/constants/codexClient.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
// refresh this so the fingerprint OpenAI sees from the OAuth/Responses face
// matches the real client version. Overridable per-deployment via
// CODEX_CLIENT_VERSION.
export const DEFAULT_CODEX_CLIENT_VERSION = "0.156.1";
export const DEFAULT_CODEX_CLIENT_VERSION = "0.159.2";
export const CODEX_CLI_RS_ORIGINATOR = "codex_cli_rs";

export function getCodexCliRsHeaders(
Expand Down
17 changes: 17 additions & 0 deletions src/shared/constants/pricing/oauth-subscriptions.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,16 @@ const GPT_6_SOL_CODEX_PRICING = {
reasoning: 10.0,
cache_creation: 2.5,
};
// GPT-6.1 Sol Standard, from models.dev openai/gpt-6.1-sol and the OpenAI
// model page. Cached input is $0.10, not the $0.20 GPT-6 Sol rate.
// https://developers.openai.com/api/docs/models/gpt-6.1-sol
const GPT_6_1_SOL_CODEX_PRICING = {
input: 2.0,
output: 10.0,
cached: 0.1,
reasoning: 10.0,
cache_creation: 2.5,
};
const GPT_6_LUNA_CODEX_PRICING = {
input: 0.1,
output: 0.5,
Expand Down Expand Up @@ -122,6 +132,13 @@ export const DEFAULT_PRICING_OAUTH = {
"gpt-6-sol-high": GPT_6_SOL_CODEX_PRICING,
"gpt-6-sol-medium": GPT_6_SOL_CODEX_PRICING,
"gpt-6-sol-low": GPT_6_SOL_CODEX_PRICING,
"gpt-6.1-sol": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6.1-sol-ultra": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6.1-sol-max": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6.1-sol-xhigh": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6.1-sol-high": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6.1-sol-medium": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6.1-sol-low": GPT_6_1_SOL_CODEX_PRICING,
"gpt-6-luna": GPT_6_LUNA_CODEX_PRICING,
"gpt-6-luna-max": GPT_6_LUNA_CODEX_PRICING,
"gpt-6-luna-xhigh": GPT_6_LUNA_CODEX_PRICING,
Expand Down
5 changes: 3 additions & 2 deletions src/shared/reasoning/effortStandardization.ts
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,7 @@ export function extendCodexGpt56EffortValues(
}

const match = normalizedModel.match(
/^gpt-(?:5\.6-(sol|terra|luna)|6-(astra|sol|luna))(?:-(?:none|low|medium|high|xhigh|max|ultra))?$/
/^gpt-(?:5\.6-(sol|terra|luna)|6(?:\.\d+)?-(astra|sol|luna))(?:-(?:none|low|medium|high|xhigh|max|ultra))?$/
);
if (!match) return values;

Expand All @@ -53,7 +53,8 @@ export function extendCodexGpt56EffortValues(
if (normalizedProvider !== "codex" && normalizedProvider !== "cx") return values;

const nativeValues = ["low", "medium", "high", "xhigh", "max"];
return (match[1] || match[2]) === "luna" ? nativeValues : [...nativeValues, "ultra"];
const family = match[1] || match[2];
return family === "luna" ? nativeValues : [...nativeValues, "ultra"];
}

/**
Expand Down
12 changes: 6 additions & 6 deletions tests/snapshots/provider/translate-path.json
Original file line number Diff line number Diff line change
Expand Up @@ -1334,23 +1334,23 @@
"Authorization": "Bearer <TOK>",
"Content-Type": "application/json",
"Openai-Beta": "responses_websockets=2026-02-06",
"User-Agent": "codex-cli/0.156.1 (<OS>; <ARCH>)",
"Version": "0.156.1"
"User-Agent": "codex-cli/0.159.2 (<OS>; <ARCH>)",
"Version": "0.159.2"
},
"nonStream": {
"Authorization": "Bearer <TOK>",
"Content-Type": "application/json",
"Openai-Beta": "responses_websockets=2026-02-06",
"User-Agent": "codex-cli/0.156.1 (<OS>; <ARCH>)",
"Version": "0.156.1"
"User-Agent": "codex-cli/0.159.2 (<OS>; <ARCH>)",
"Version": "0.159.2"
},
"oauth": {
"Accept": "text/event-stream",
"Authorization": "Bearer <TOK>",
"Content-Type": "application/json",
"Openai-Beta": "responses_websockets=2026-02-06",
"User-Agent": "codex-cli/0.156.1 (<OS>; <ARCH>)",
"Version": "0.156.1"
"User-Agent": "codex-cli/0.159.2 (<OS>; <ARCH>)",
"Version": "0.159.2"
}
},
"url": {
Expand Down
9 changes: 5 additions & 4 deletions tests/unit/8951-github-gpt56-responses.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,14 +6,15 @@ import { GithubExecutor } from "../../open-sse/executors/github.ts";
test("#8951 GitHub GPT-5.6 models must use the Responses endpoint", () => {
const executor = new GithubExecutor();
const urls = Object.fromEntries(
["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"].map((model) => [
model,
executor.buildUrl(model, false),
])
["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", "gpt-6-astra", "gpt-6.1-sol"].map(
(model) => [model, executor.buildUrl(model, false)]
)
);
assert.deepEqual(urls, {
"gpt-5.6-sol": "https://api.githubcopilot.com/responses",
"gpt-5.6-terra": "https://api.githubcopilot.com/responses",
"gpt-5.6-luna": "https://api.githubcopilot.com/responses",
"gpt-6-astra": "https://api.githubcopilot.com/responses",
"gpt-6.1-sol": "https://api.githubcopilot.com/responses",
});
});
1 change: 1 addition & 0 deletions tests/unit/codex-fast-tier.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,7 @@ test("Codex global service mode distinguishes no setting from explicit tiers", (
supportedModels: [
"gpt-6-astra",
"gpt-6-sol",
"gpt-6.1-sol",
"gpt-6-luna",
"gpt-5.6-sol",
"gpt-5.6-terra",
Expand Down
41 changes: 40 additions & 1 deletion tests/unit/codex-gpt6-sol-luna.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,14 +5,15 @@ import { getModelsByProviderId } from "../../open-sse/config/providerModels.ts";
import { CodexExecutor } from "../../open-sse/executors/codex.ts";
import { openaiToOpenAIResponsesRequest } from "../../open-sse/translator/request/openai-responses/toResponses.ts";
import { getPricingForModel } from "../../src/shared/constants/pricing.ts";
import { getCodexFastCostMultiplier } from "../../src/lib/usage/costCalculator.ts";
import { computeCostFromPricing, getCodexFastCostMultiplier } from "../../src/lib/usage/costCalculator.ts";
import { extendCodexGpt56EffortValues } from "../../src/shared/reasoning/effortStandardization.ts";
import * as reasoningMetadata from "../../src/lib/vscode/reasoningMetadata.ts";

// The live Codex catalog (client 0.155.1) advertises Sol at low..ultra and Luna at
// low..max, mirroring the GPT-5.6 Sol/Luna split.
const FAMILIES = [
{ model: "gpt-6-sol", efforts: ["ultra", "max", "xhigh", "high", "medium", "low"] },
{ model: "gpt-6.1-sol", efforts: ["ultra", "max", "xhigh", "high", "medium", "low"] },
{ model: "gpt-6-luna", efforts: ["max", "xhigh", "high", "medium", "low"] },
] as const;

Expand Down Expand Up @@ -112,6 +113,10 @@ test("Catalog effort tiers follow the live Codex levels for GPT-6 models", () =>
for (const provider of ["codex", "cx"]) {
assert.deepEqual(extendCodexGpt56EffortValues(provider, "gpt-6-astra", base), withUltra);
assert.deepEqual(extendCodexGpt56EffortValues(provider, "gpt-6-sol", base), withUltra);
assert.deepEqual(extendCodexGpt56EffortValues(provider, "gpt-6.1-sol", base), withUltra);
// A later minor of the same family keeps ultra without a new allowlist entry.
assert.deepEqual(extendCodexGpt56EffortValues(provider, "gpt-6.2-sol", base), withUltra);
assert.deepEqual(extendCodexGpt56EffortValues(provider, "gpt-6.1-luna", base), withMax);
assert.deepEqual(extendCodexGpt56EffortValues(provider, "gpt-6-luna-max", base), withMax);
}
// Kiro's GPT-5.6 max extension does not reach models Kiro does not serve.
Expand Down Expand Up @@ -179,6 +184,7 @@ test("GPT-6 Sol and Luna Codex pricing and Fast multiplier match the credit rate
// Sol 50 / 5 / 250, Luna 2.5 / 0.25 / 12.5. Fast is 2.5x Standard for both.
const expected = {
"gpt-6-sol": { input: 2, cached: 0.2, output: 10 },
"gpt-6.1-sol": { input: 2, cached: 0.1, output: 10 },
"gpt-6-luna": { input: 0.1, cached: 0.01, output: 0.5 },
};
for (const { model, efforts } of FAMILIES) {
Expand All @@ -196,3 +202,36 @@ test("GPT-6 Sol and Luna Codex pricing and Fast multiplier match the credit rate
}
}
});

test("GPT-6.1 Sol bills the whole request at the long-context rate past 272K input", () => {
const pricing = { input: 2, cached: 0.1, output: 10, cache_creation: 2.5, reasoning: 10 };
const tokens = { input: 300_000, cache_read_input_tokens: 50_000, output: 1_000 };
const standard = computeCostFromPricing(
pricing,
{ input: 272_000, output: 1_000 },
{ model: "gpt-6.1-sol" }
);
const elevated = computeCostFromPricing(
pricing,
{ ...tokens, reasoning_tokens: 200 },
{ model: "gpt-6.1-sol-high" }
);
const olderSol = computeCostFromPricing(pricing, tokens, { model: "gpt-6-sol" });

const standardExpected = (272_000 * 2 + 1_000 * 10) / 1_000_000;
// Output is already billed at 1.5x, which is above the reasoning price.
// The premium cannot go negative or the request would be under-charged.
const elevatedExpected = (250_000 * 4 + 50_000 * 0.2 + 1_000 * 15) / 1_000_000;
const olderExpected = (250_000 * 2 + 50_000 * 0.1 + 1_000 * 10) / 1_000_000;
assert.ok(Math.abs(standard - standardExpected) < 1e-9);
assert.ok(Math.abs(elevated - elevatedExpected) < 1e-9);
assert.ok(Math.abs(olderSol - olderExpected) < 1e-9);
});

test("GPT-6.1 Sol stays at the standard rate one token under 272K and doubles one token over", () => {
const pricing = { input: 2, output: 10 };
const under = computeCostFromPricing(pricing, { input: 271_999, output: 0 }, { model: "gpt-6.1-sol" });
const over = computeCostFromPricing(pricing, { input: 272_001, output: 0 }, { model: "gpt-6.1-sol" });
assert.ok(Math.abs(under - (271_999 * 2) / 1_000_000) < 1e-9);
assert.ok(Math.abs(over - (272_001 * 4) / 1_000_000) < 1e-9);
});
6 changes: 3 additions & 3 deletions tests/unit/executor-codex.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -184,10 +184,10 @@ test("CodexExecutor.buildHeaders binds workspace ids and disables SSE accept for
assert.equal(standardHeaders.Authorization, "Bearer codex-token");
assert.equal(standardHeaders.Accept, "text/event-stream");
assert.equal(standardHeaders["chatgpt-account-id"], "workspace-1");
assert.equal(standardHeaders.Version, "0.156.1");
assert.equal(standardHeaders.Version, "0.159.2");
assert.equal(standardHeaders["Openai-Beta"], "responses_websockets=2026-02-06");
assert.equal(standardHeaders["X-Codex-Beta-Features"], undefined);
assert.equal(standardHeaders["User-Agent"], "codex-cli/0.156.1 (Windows 10.0.26200; x64)");
assert.equal(standardHeaders["User-Agent"], "codex-cli/0.159.2 (Windows 10.0.26200; x64)");
assert.equal(compactHeaders.Accept, "application/json");
});

Expand All @@ -213,7 +213,7 @@ test("CodexExecutor.buildHeaders honors safe env overrides for Version and User-
},
() => {
const headers = executor.buildHeaders({ accessToken: "codex-token" }, true);
assert.equal(headers.Version, "0.156.1");
assert.equal(headers.Version, "0.159.2");
assert.equal(headers["User-Agent"], "custom-codex/9.9.9");
}
);
Expand Down
Loading