Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions core/providers/anthropic/chat.go
Original file line number Diff line number Diff line change
Expand Up @@ -984,6 +984,10 @@ func (response *AnthropicMessageResponse) ToBifrostChatResponse(ctx *schemas.Bif
mapped := MapAnthropicServiceTierToBifrost(*response.Usage.ServiceTier)
bifrostResponse.ServiceTier = &mapped
}
// Forward the speed actually served (fast mode) — drives fast-mode billing.
if response.Usage.Speed != nil {
bifrostResponse.Speed = response.Usage.Speed
}
}

return bifrostResponse
Expand Down Expand Up @@ -1031,6 +1035,10 @@ func ToAnthropicChatResponse(bifrostResp *schemas.BifrostChatResponse) *Anthropi
mapped := MapBifrostServiceTierToAnthropicResponse(*bifrostResp.ServiceTier)
anthropicResp.Usage.ServiceTier = &mapped
}
// Forward the speed actually served (fast mode)
if bifrostResp.Speed != nil {
anthropicResp.Usage.Speed = bifrostResp.Speed
}
Comment thread
TejasGhatte marked this conversation as resolved.
}

// Convert choices to content
Expand Down
6 changes: 6 additions & 0 deletions core/providers/anthropic/passthrough_usage.go
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,9 @@ func buildAnthropicPassthroughUsage(au *AnthropicUsage) *schemas.BifrostPassthro
t := MapAnthropicServiceTierToBifrost(*au.ServiceTier)
u.ServiceTier = &t
}
if au.Speed != nil {
u.Speed = au.Speed
}
return u
}

Expand Down Expand Up @@ -136,6 +139,9 @@ func (a *AnthropicPassthroughStreamUsage) ObserveEvent(event []byte) *schemas.Bi
if u.ServiceTier != nil {
c.ServiceTier = u.ServiceTier
}
if u.Speed != nil {
c.Speed = u.Speed
}
return a.usage()
}

Expand Down
12 changes: 12 additions & 0 deletions core/providers/anthropic/responses.go
Original file line number Diff line number Diff line change
Expand Up @@ -2866,6 +2866,11 @@ func (response *AnthropicMessageResponse) ToBifrostResponsesResponse(ctx *schema
bifrostResp.ServiceTier = &mapped
}

// Forward the speed actually served (fast mode) — drives fast-mode billing.
if response.Usage != nil && response.Usage.Speed != nil {
bifrostResp.Speed = response.Usage.Speed
}

return bifrostResp
}

Expand Down Expand Up @@ -2928,6 +2933,13 @@ func ToAnthropicResponsesResponse(ctx *schemas.BifrostContext, bifrostResp *sche
anthropicResp.Usage.ServiceTier = &mapped
}

if bifrostResp.Speed != nil {
if anthropicResp.Usage == nil {
anthropicResp.Usage = &AnthropicUsage{}
}
anthropicResp.Usage.Speed = bifrostResp.Speed
}

return anthropicResp
}

Expand Down
1 change: 1 addition & 0 deletions core/providers/anthropic/types.go
Original file line number Diff line number Diff line change
Expand Up @@ -1429,6 +1429,7 @@ type AnthropicUsage struct {
OutputTokens int `json:"output_tokens"`
ServerToolUse *AnthropicServerToolUseUsage `json:"server_tool_use,omitempty"` // Server tool use statistics (e.g., web search)
ServiceTier *string `json:"service_tier,omitempty"` // "standard", "priority", or "batch"
Speed *string `json:"speed,omitempty"` // "fast" or "standard" — which speed was actually served (fast mode research preview)
InferenceGeo *string `json:"inference_geo,omitempty"` // the geographic region for inference processing. If not specified, the workspace's default_inference_geo is used.
Iterations []AnthropicUsage `json:"iterations,omitempty"` // Iterations statistics
}
Expand Down
2 changes: 1 addition & 1 deletion core/providers/anthropic/utils.go
Original file line number Diff line number Diff line change
Expand Up @@ -1079,7 +1079,7 @@ func AddMissingBetaHeadersToContext(ctx *schemas.BifrostContext, req *AnthropicM
// Check for fast mode. Only add the beta header when both the provider
// supports fast mode AND the model does (Opus 4.6 only per
// SupportsFastMode); otherwise sending the header guarantees a 400.
if req.Speed != nil && *req.Speed == "fast" {
if req.Speed != nil {
if (!hasProvider || features.FastMode) && SupportsFastMode(req.Model) {
headers = appendUniqueHeader(headers, AnthropicFastModeBetaHeader)
}
Expand Down
1 change: 1 addition & 0 deletions core/schemas/chatcompletions.go
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@ type BifrostChatResponse struct {
Model string `json:"model"`
Object string `json:"object"` // "chat.completion" or "chat.completion.chunk"
ServiceTier *BifrostServiceTier `json:"service_tier,omitempty"`
Speed *string `json:"speed,omitempty"` // "fast" | "standard" — speed actually served (Anthropic fast mode); drives fast-mode billing
SystemFingerprint string `json:"system_fingerprint"`
Usage *BifrostLLMUsage `json:"usage"`
ExtraFields BifrostResponseExtraFields `json:"extra_fields"`
Expand Down
1 change: 1 addition & 0 deletions core/schemas/passthrough.go
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ type BifrostPassthroughUsage struct {
// Text / chat / responses / embeddings
LLMUsage *BifrostLLMUsage
ServiceTier *BifrostServiceTier // "priority" | "flex" | nil (default)
Speed *string // "fast" | "standard" — speed actually served (Anthropic fast mode); drives fast-mode billing

// Image generation / edit / variation
ImageUsage *ImageUsage
Expand Down
1 change: 1 addition & 0 deletions core/schemas/responses.go
Original file line number Diff line number Diff line change
Expand Up @@ -129,6 +129,7 @@ type BifrostResponsesResponse struct {
Reasoning *ResponsesParametersReasoning `json:"reasoning"` // Configuration options for reasoning models
SafetyIdentifier *string `json:"safety_identifier"` // Safety identifier
ServiceTier *BifrostServiceTier `json:"service_tier"`
Speed *string `json:"speed,omitempty"` // "fast" | "standard" — speed actually served (Anthropic fast mode); drives fast-mode billing
Comment thread
TejasGhatte marked this conversation as resolved.
Status *string `json:"status,omitempty"` // completed, failed, in_progress, cancelled, queued, or incomplete
StreamOptions *ResponsesStreamOptions `json:"stream_options,omitempty"`
StopReason *string `json:"stop_reason,omitempty"` // Not in OpenAI's spec, but sent by other providers
Expand Down
52 changes: 51 additions & 1 deletion framework/configstore/migrations.go
Original file line number Diff line number Diff line change
Expand Up @@ -188,7 +188,6 @@ type legacyBudgetTeam struct {
// TableName returns the governance_teams table name for legacyBudgetTeam.
func (legacyBudgetTeam) TableName() string { return "governance_teams" }


// sqliteColumnInfo holds the information about a SQLite column.
type sqliteColumnInfo struct {
Name string `gorm:"column:name"`
Expand Down Expand Up @@ -875,6 +874,9 @@ func triggerMigrations(ctx context.Context, db *gorm.DB) error {
if err := migrationAddMCPLibrarySourceColumns(ctx, db); err != nil {
return err
}
if err := migrationAddFastModePricingColumns(ctx, db); err != nil {
return err
}
return nil
}

Expand Down Expand Up @@ -7829,6 +7831,54 @@ func migrationAddFlexTierPricingColumns(ctx context.Context, db *gorm.DB) error
return nil
}

// migrationAddFastModePricingColumns adds pricing columns for Anthropic fast mode
// (research preview, speed:"fast" on Opus 4.6/4.7/4.8).
func migrationAddFastModePricingColumns(ctx context.Context, db *gorm.DB) error {
m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{
ID: "add_fast_mode_pricing_columns",
Migrate: func(tx *gorm.DB) error {
tx = tx.WithContext(ctx)
mg := tx.Migrator()

columns := []string{
"input_cost_per_token_fast",
"output_cost_per_token_fast",
}

for _, field := range columns {
if !mg.HasColumn(&tables.TableModelPricing{}, field) {
if err := mg.AddColumn(&tables.TableModelPricing{}, field); err != nil {
return fmt.Errorf("failed to add column %s: %w", field, err)
}
}
}
return nil
},
Rollback: func(tx *gorm.DB) error {
tx = tx.WithContext(ctx)
mg := tx.Migrator()

columns := []string{
"input_cost_per_token_fast",
"output_cost_per_token_fast",
}

for _, field := range columns {
if mg.HasColumn(&tables.TableModelPricing{}, field) {
if err := mg.DropColumn(&tables.TableModelPricing{}, field); err != nil {
return fmt.Errorf("failed to drop column %s: %w", field, err)
}
}
}
return nil
},
}})
if err := m.Migrate(); err != nil {
return fmt.Errorf("error while running fast mode pricing columns migration: %s", err.Error())
}
return nil
}

// migrationAddWhitelistedRoutesJSONColumn adds the whitelisted_routes_json column to the config_client table
func migrationAddWhitelistedRoutesJSONColumn(ctx context.Context, db *gorm.DB) error {
m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{
Expand Down
2 changes: 2 additions & 0 deletions framework/configstore/rdb.go
Original file line number Diff line number Diff line change
Expand Up @@ -2332,6 +2332,8 @@ var pricingSyncUpdateColumns = []string{
"output_cost_per_token_priority",
"input_cost_per_token_flex",
"output_cost_per_token_flex",
"input_cost_per_token_fast",
"output_cost_per_token_fast",
"input_cost_per_character",
// Costs - 128k Tier
"input_cost_per_token_above_128k_tokens",
Expand Down
6 changes: 5 additions & 1 deletion framework/configstore/tables/modelpricing.go
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,11 @@ type TableModelPricing struct {
OutputCostPerTokenPriority *float64 `gorm:"default:null;column:output_cost_per_token_priority" json:"output_cost_per_token_priority,omitempty"`
InputCostPerTokenFlex *float64 `gorm:"default:null;column:input_cost_per_token_flex" json:"input_cost_per_token_flex,omitempty"`
OutputCostPerTokenFlex *float64 `gorm:"default:null;column:output_cost_per_token_flex" json:"output_cost_per_token_flex,omitempty"`
InputCostPerCharacter *float64 `gorm:"default:null;column:input_cost_per_character" json:"input_cost_per_character,omitempty"`
// Fast mode (Anthropic research preview, speed:"fast" on Opus 4.6/4.7/4.8).
// Flat rate across the full context window; cache tokens bill at standard cache rates.
InputCostPerTokenFast *float64 `gorm:"default:null;column:input_cost_per_token_fast" json:"input_cost_per_token_fast,omitempty"`
OutputCostPerTokenFast *float64 `gorm:"default:null;column:output_cost_per_token_fast" json:"output_cost_per_token_fast,omitempty"`
InputCostPerCharacter *float64 `gorm:"default:null;column:input_cost_per_character" json:"input_cost_per_character,omitempty"`
// Costs - 128k Tier
InputCostPerTokenAbove128kTokens *float64 `gorm:"default:null;column:input_cost_per_token_above_128k_tokens" json:"input_cost_per_token_above_128k_tokens,omitempty"`
InputCostPerImageAbove128kTokens *float64 `gorm:"default:null;column:input_cost_per_image_above_128k_tokens" json:"input_cost_per_image_above_128k_tokens,omitempty"`
Expand Down
48 changes: 30 additions & 18 deletions framework/modelcatalog/datasheet/cost.go
Original file line number Diff line number Diff line change
Expand Up @@ -187,18 +187,18 @@ func extractCostInput(result *schemas.BifrostResponse) costInput {

case result.ChatResponse != nil && result.ChatResponse.Usage != nil:
input.usage = result.ChatResponse.Usage
input.tier = tierFromString(result.ChatResponse.ServiceTier)
input.tier = tierFromResponse(result.ChatResponse.ServiceTier, result.ChatResponse.Speed)

case result.ResponsesResponse != nil && result.ResponsesResponse.Usage != nil:
input.usage = responsesUsageToBifrostUsage(result.ResponsesResponse.Usage)
input.tier = tierFromString(result.ResponsesResponse.ServiceTier)
input.tier = tierFromResponse(result.ResponsesResponse.ServiceTier, result.ResponsesResponse.Speed)

case result.CompactionResponse != nil && result.CompactionResponse.Usage != nil:
input.usage = responsesUsageToBifrostUsage(result.CompactionResponse.Usage)

case result.ResponsesStreamResponse != nil && result.ResponsesStreamResponse.Response != nil && result.ResponsesStreamResponse.Response.Usage != nil:
input.usage = responsesUsageToBifrostUsage(result.ResponsesStreamResponse.Response.Usage)
input.tier = tierFromString(result.ResponsesStreamResponse.Response.ServiceTier)
input.tier = tierFromResponse(result.ResponsesStreamResponse.Response.ServiceTier, result.ResponsesStreamResponse.Response.Speed)

case result.EmbeddingResponse != nil && result.EmbeddingResponse.Usage != nil:
input.usage = result.EmbeddingResponse.Usage
Expand Down Expand Up @@ -730,24 +730,33 @@ func computeOCRCost(pricing *configstoreTables.TableModelPricing, ocrProcessedPa
// Helpers
// ---------------------------------------------------------------------------

// tierFromString constructs a serviceTier from an OpenAI service_tier response value.
func tierFromString(s *schemas.BifrostServiceTier) serviceTier {
if s == nil {
return serviceTier{}
}
switch *s {
case schemas.BifrostServiceTierPriority:
return serviceTier{isPriority: true}
case schemas.BifrostServiceTierFlex:
return serviceTier{isFlex: true}
default:
return serviceTier{}
// tierFromResponse builds a serviceTier from a response's billing-relevant
// fields: the OpenAI service_tier (priority/flex) and the Anthropic speed
// (fast mode). speed == "fast" means fast mode was actually served — the
// provider echoes the served speed, so stripped/fell-back requests report
// "standard" and bill at standard rates.
func tierFromResponse(s *schemas.BifrostServiceTier, speed *string) serviceTier {
var tier serviceTier
if s != nil {
switch *s {
case schemas.BifrostServiceTierPriority:
tier.isPriority = true
case schemas.BifrostServiceTierFlex:
tier.isFlex = true
}
}
tier.isFast = speed != nil && *speed == "fast"
return tier
}

// tieredInputRate returns the effective per-token input rate based on total token count.
// Flex applies a flat rate. Priority-specific tier rates are preferred where available.
func tieredInputRate(pricing *configstoreTables.TableModelPricing, totalTokens int, tier serviceTier) float64 {
// Fast mode (Anthropic) is a flat rate across the full context window — it
// takes precedence over the token-count tiers below.
if tier.isFast && pricing.InputCostPerTokenFast != nil {
return *pricing.InputCostPerTokenFast
}
Comment thread
TejasGhatte marked this conversation as resolved.
if tier.isFlex && pricing.InputCostPerTokenFlex != nil {
return *pricing.InputCostPerTokenFlex
}
Expand Down Expand Up @@ -782,6 +791,11 @@ func tieredInputRate(pricing *configstoreTables.TableModelPricing, totalTokens i
// tieredOutputRate returns the effective per-token output rate based on total token count.
// Flex applies a flat rate. Priority-specific tier rates are preferred where available.
func tieredOutputRate(pricing *configstoreTables.TableModelPricing, totalTokens int, tier serviceTier) float64 {
// Fast mode (Anthropic) is a flat rate across the full context window — it
// takes precedence over the token-count tiers below.
if tier.isFast && pricing.OutputCostPerTokenFast != nil {
return *pricing.OutputCostPerTokenFast
}
if tier.isFlex && pricing.OutputCostPerTokenFlex != nil {
return *pricing.OutputCostPerTokenFlex
}
Expand Down Expand Up @@ -1298,9 +1312,7 @@ func passthroughUsageToCostInput(su *schemas.BifrostPassthroughUsage) costInput
if su.LLMUsage != nil {
input.usage = su.LLMUsage
}
if su.ServiceTier != nil {
input.tier = tierFromString(su.ServiceTier)
}
input.tier = tierFromResponse(su.ServiceTier, su.Speed)
if su.ImageUsage != nil {
input.imageUsage = su.ImageUsage
input.imageSize = su.ImageSize
Expand Down
Loading
Loading