Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@

### Changed
- Menu: move each usage window's used percentage and reset time into its title row, with all pace detail on one line (#2182). Thanks @jack24254029!
- Codex: define Fast cost as estimated API Fast USD, resolve it models.dev-first with model-specific API ratios, and refresh GPT-5.6 Terra/Luna fallback rates (refs #2175). Thanks @iam-brain!

### Fixed
- Sync: propagate provider configuration edits made by the CLI or directly in `config.json` to the iCloud fleet without echoing remotely applied writes.
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
// Generated by Scripts/regenerate-codex-parser-hash.sh. Do not edit by hand.

enum CodexParserHash {
static let value = "a72389ecaa16bc9a"
static let value = "843ca061c36bbea1"
}
126 changes: 49 additions & 77 deletions Sources/CodexBarCore/Vendored/CostUsage/CostUsagePricing.swift
Original file line number Diff line number Diff line change
Expand Up @@ -18,10 +18,6 @@ enum CostUsagePricing {
let outputCostPerTokenAboveThreshold: Double?
let cacheReadInputCostPerTokenAboveThreshold: Double?
let cacheWriteInputCostPerTokenAboveThreshold: Double?
let priorityInputCostPerToken: Double?
let priorityOutputCostPerToken: Double?
let priorityCacheReadInputCostPerToken: Double?
let priorityCacheWriteInputCostPerToken: Double?

init(
inputCostPerToken: Double,
Expand All @@ -33,11 +29,7 @@ enum CostUsagePricing {
inputCostPerTokenAboveThreshold: Double? = nil,
outputCostPerTokenAboveThreshold: Double? = nil,
cacheReadInputCostPerTokenAboveThreshold: Double? = nil,
cacheWriteInputCostPerTokenAboveThreshold: Double? = nil,
priorityInputCostPerToken: Double? = nil,
priorityOutputCostPerToken: Double? = nil,
priorityCacheReadInputCostPerToken: Double? = nil,
priorityCacheWriteInputCostPerToken: Double? = nil)
cacheWriteInputCostPerTokenAboveThreshold: Double? = nil)
{
self.inputCostPerToken = inputCostPerToken
self.outputCostPerToken = outputCostPerToken
Expand All @@ -49,10 +41,6 @@ enum CostUsagePricing {
self.outputCostPerTokenAboveThreshold = outputCostPerTokenAboveThreshold
self.cacheReadInputCostPerTokenAboveThreshold = cacheReadInputCostPerTokenAboveThreshold
self.cacheWriteInputCostPerTokenAboveThreshold = cacheWriteInputCostPerTokenAboveThreshold
self.priorityInputCostPerToken = priorityInputCostPerToken
self.priorityOutputCostPerToken = priorityOutputCostPerToken
self.priorityCacheReadInputCostPerToken = priorityCacheReadInputCostPerToken
self.priorityCacheWriteInputCostPerToken = priorityCacheWriteInputCostPerToken
}
}

Expand Down Expand Up @@ -156,18 +144,12 @@ enum CostUsagePricing {
thresholdTokens: 272_000,
inputCostPerTokenAboveThreshold: 5e-6,
outputCostPerTokenAboveThreshold: 2.25e-5,
cacheReadInputCostPerTokenAboveThreshold: 5e-7,
priorityInputCostPerToken: 5e-6,
priorityOutputCostPerToken: 3e-5,
priorityCacheReadInputCostPerToken: 5e-7),
cacheReadInputCostPerTokenAboveThreshold: 5e-7),
"gpt-5.4-mini": CodexPricing(
inputCostPerToken: 7.5e-7,
outputCostPerToken: 4.5e-6,
cacheReadInputCostPerToken: 7.5e-8,
displayLabel: nil,
priorityInputCostPerToken: 1.5e-6,
priorityOutputCostPerToken: 9e-6,
priorityCacheReadInputCostPerToken: 1.5e-7),
displayLabel: nil),
"gpt-5.4-nano": CodexPricing(
inputCostPerToken: 2e-7,
outputCostPerToken: 1.25e-6,
Expand All @@ -186,19 +168,16 @@ enum CostUsagePricing {
thresholdTokens: 272_000,
inputCostPerTokenAboveThreshold: 1e-5,
outputCostPerTokenAboveThreshold: 4.5e-5,
cacheReadInputCostPerTokenAboveThreshold: 1e-6,
priorityInputCostPerToken: 1.25e-5,
priorityOutputCostPerToken: 7.5e-5,
priorityCacheReadInputCostPerToken: 1.25e-6),
cacheReadInputCostPerTokenAboveThreshold: 1e-6),
"gpt-5.5-pro": CodexPricing(
inputCostPerToken: 3e-5,
outputCostPerToken: 1.8e-4,
cacheReadInputCostPerToken: nil,
displayLabel: nil),
// GPT-5.6 Sol/Terra/Luna (OpenAI pricing page + model cards).
// Long context: prompts with >272K input tokens are 2x input / 1.5x output for the full
// request. Cache writes: 1.25x uncached input. Priority rates are explicit because support
// and multipliers are provider contracts, not properties that can be inferred from Standard.
// request. Cache writes: 1.25x uncached input. API Fast support and multipliers are applied
// separately after Standard pricing resolves from models.dev or this bundled fallback.
"gpt-5.6-sol": CodexPricing(
inputCostPerToken: 5e-6,
outputCostPerToken: 3e-5,
Expand All @@ -209,45 +188,36 @@ enum CostUsagePricing {
inputCostPerTokenAboveThreshold: 1e-5,
outputCostPerTokenAboveThreshold: 4.5e-5,
cacheReadInputCostPerTokenAboveThreshold: 1e-6,
cacheWriteInputCostPerTokenAboveThreshold: 1.25e-5,
priorityInputCostPerToken: 1e-5,
priorityOutputCostPerToken: 6e-5,
priorityCacheReadInputCostPerToken: 1e-6,
priorityCacheWriteInputCostPerToken: 1.25e-5),
cacheWriteInputCostPerTokenAboveThreshold: 1.25e-5),
"gpt-5.6-terra": CodexPricing(
inputCostPerToken: 2.5e-6,
outputCostPerToken: 1.5e-5,
cacheReadInputCostPerToken: 2.5e-7,
inputCostPerToken: 2e-6,
outputCostPerToken: 1.2e-5,
cacheReadInputCostPerToken: 2e-7,
displayLabel: nil,
cacheWriteInputCostPerToken: 3.125e-6,
cacheWriteInputCostPerToken: 2.5e-6,
thresholdTokens: 272_000,
inputCostPerTokenAboveThreshold: 5e-6,
outputCostPerTokenAboveThreshold: 2.25e-5,
cacheReadInputCostPerTokenAboveThreshold: 5e-7,
cacheWriteInputCostPerTokenAboveThreshold: 6.25e-6,
priorityInputCostPerToken: 5e-6,
priorityOutputCostPerToken: 3e-5,
priorityCacheReadInputCostPerToken: 5e-7,
priorityCacheWriteInputCostPerToken: 6.25e-6),
inputCostPerTokenAboveThreshold: 4e-6,
outputCostPerTokenAboveThreshold: 1.8e-5,
cacheReadInputCostPerTokenAboveThreshold: 4e-7,
cacheWriteInputCostPerTokenAboveThreshold: 5e-6),
"gpt-5.6-luna": CodexPricing(
inputCostPerToken: 1e-6,
outputCostPerToken: 6e-6,
cacheReadInputCostPerToken: 1e-7,
inputCostPerToken: 2e-7,
outputCostPerToken: 1.2e-6,
cacheReadInputCostPerToken: 2e-8,
displayLabel: nil,
cacheWriteInputCostPerToken: 1.25e-6,
cacheWriteInputCostPerToken: 2.5e-7,
thresholdTokens: 272_000,
inputCostPerTokenAboveThreshold: 2e-6,
outputCostPerTokenAboveThreshold: 9e-6,
cacheReadInputCostPerTokenAboveThreshold: 2e-7,
cacheWriteInputCostPerTokenAboveThreshold: 2.5e-6,
priorityInputCostPerToken: 2e-6,
priorityOutputCostPerToken: 1.2e-5,
priorityCacheReadInputCostPerToken: 2e-7,
priorityCacheWriteInputCostPerToken: 2.5e-6),
inputCostPerTokenAboveThreshold: 4e-7,
outputCostPerTokenAboveThreshold: 1.8e-6,
cacheReadInputCostPerTokenAboveThreshold: 4e-8,
cacheWriteInputCostPerTokenAboveThreshold: 5e-7),
]

static func codexBuiltInPricingFingerprint() -> String {
var parts = ["priorityInputTokenLimit=\(self.codexPriorityInputTokenLimit)"]
var parts = [
"priorityInputTokenLimit=\(self.codexPriorityInputTokenLimit)",
"fastPricingDefinition=api-fast-usd-v1",
]
for model in self.codex.keys.sorted() {
guard let pricing = self.codex[model] else { continue }
parts.append([
Expand All @@ -262,10 +232,7 @@ enum CostUsagePricing {
self.optionalPricingFingerprint(pricing.outputCostPerTokenAboveThreshold),
self.optionalPricingFingerprint(pricing.cacheReadInputCostPerTokenAboveThreshold),
self.optionalPricingFingerprint(pricing.cacheWriteInputCostPerTokenAboveThreshold),
self.optionalPricingFingerprint(pricing.priorityInputCostPerToken),
self.optionalPricingFingerprint(pricing.priorityOutputCostPerToken),
self.optionalPricingFingerprint(pricing.priorityCacheReadInputCostPerToken),
self.optionalPricingFingerprint(pricing.priorityCacheWriteInputCostPerToken),
self.optionalPricingFingerprint(self.codexAPIFastMultiplier(model: model)),
].joined(separator: "|"))
}
return parts.joined(separator: "\n")
Expand Down Expand Up @@ -590,31 +557,36 @@ enum CostUsagePricing {
inputTokens: Int,
cachedInputTokens: Int = 0,
cacheWriteInputTokens: Int = 0,
outputTokens: Int) -> Double?
outputTokens: Int,
modelsDevCatalog: ModelsDevCatalog? = nil,
modelsDevCacheRoot: URL? = nil) -> Double?
{
let key = self.normalizeCodexModel(model)
guard let pricing = self.codex[key],
let priorityInputCostPerToken = pricing.priorityInputCostPerToken,
let priorityOutputCostPerToken = pricing.priorityOutputCostPerToken
else { return nil }
// OpenAI does not support Priority processing for long-context requests. Do not combine
// the independent Standard long-context and Priority short-context rate tables.
guard let multiplier = self.codexAPIFastMultiplier(model: model) else { return nil }

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Classify fast-tier traces before pricing them

This new API Fast pricing only runs after a turn is present in priorityTurns, but the trace resolver still only accepts service_tier == "priority" and only filters for the legacy priority marker. For Codex/API requests logged with the renamed service_tier: "fast" accepted by the Fast mode API docs (https://developers.openai.com/api/docs/guides/fast-mode), those turns are never marked as Fast, so they remain in the standard bucket and the multiplier here is skipped; please accept both priority and fast when building the metadata.

Useful? React with 👍 / 👎.

// OpenAI does not support API Fast processing for long-context requests. Do not combine
// the independent Standard long-context and Fast short-context rate tables.
if max(0, inputTokens) > self.codexPriorityInputTokenLimit {
return nil
}

let priorityPricing = CodexPricing(
inputCostPerToken: priorityInputCostPerToken,
outputCostPerToken: priorityOutputCostPerToken,
cacheReadInputCostPerToken: pricing.priorityCacheReadInputCostPerToken,
displayLabel: nil,
cacheWriteInputCostPerToken: pricing.priorityCacheWriteInputCostPerToken)
return self.codexCostUSD(
pricing: priorityPricing,
model: model,
inputTokens: inputTokens,
cachedInputTokens: cachedInputTokens,
outputTokens: outputTokens,
cacheWriteInputTokens: cacheWriteInputTokens,
outputTokens: outputTokens)
modelsDevCatalog: modelsDevCatalog,
modelsDevCacheRoot: modelsDevCacheRoot)
.map { $0 * multiplier }
}

/// Current public API Fast rates normalized against Standard API pricing. These are deliberately
/// distinct from ChatGPT/Codex Fast credit multipliers, which do not represent a USD charge.
static func codexAPIFastMultiplier(model: String) -> Double? {
switch self.normalizeCodexModel(model) {
case "gpt-5.4", "gpt-5.4-mini", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna": 2
case "gpt-5.5": 2.5
default: nil
}
}

private static func codexCostUSD(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -170,7 +170,9 @@ extension CostUsageScanner {
model: pricedModel,
inputTokens: row.input,
cachedInputTokens: row.cached,
outputTokens: row.output)
outputTokens: row.output,
modelsDevCatalog: modelsDevCatalog,
modelsDevCacheRoot: modelsDevCacheRoot)
else { continue }
total += max(priorityCost - baseCost, 0)
seen = true
Expand All @@ -183,11 +185,7 @@ extension CostUsageScanner {
priorityMetadata: CodexPriorityTurnMetadata) -> String
{
guard let model = priorityMetadata.model,
CostUsagePricing.codexPriorityCostUSD(
model: model,
inputTokens: row.input,
cachedInputTokens: row.cached,
outputTokens: row.output) != nil
CostUsagePricing.codexAPIFastMultiplier(model: model) != nil
else { return row.model }
return model
}
Expand Down Expand Up @@ -256,7 +254,9 @@ extension CostUsageScanner {
model: pricedModel,
inputTokens: row.input,
cachedInputTokens: row.cached,
outputTokens: row.output)
outputTokens: row.output,
modelsDevCatalog: modelsDevCatalog,
modelsDevCacheRoot: modelsDevCacheRoot)
{
breakdown.priorityCostUSD += max(priorityCost, baseCost ?? priorityCost)
breakdown.sawPriorityCost = true
Expand Down Expand Up @@ -478,7 +478,9 @@ extension CostUsageScanner {
model: pricedModel,
inputTokens: row.input,
cachedInputTokens: row.cached,
outputTokens: row.output)
outputTokens: row.output,
modelsDevCatalog: modelsDevCatalog,
modelsDevCacheRoot: modelsDevCacheRoot)
{
max(priorityCost, baseCost ?? priorityCost)
} else {
Expand Down Expand Up @@ -621,7 +623,9 @@ extension CostUsageScanner {
model: pricedModel,
inputTokens: row.input,
cachedInputTokens: row.cached,
outputTokens: row.output)
outputTokens: row.output,
modelsDevCatalog: modelsDevCatalog,
modelsDevCacheRoot: modelsDevCacheRoot)
{
priorityCostNanos[row.day, default: [:]][row.model, default: 0] += Int64(
(max(priorityCost, baseCost ?? priorityCost) * Self.costScale).rounded())
Expand Down
Loading