Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions core/bifrost.go
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,7 @@ import (
"github.com/maximhq/bifrost/core/providers/replicate"
"github.com/maximhq/bifrost/core/providers/runware"
"github.com/maximhq/bifrost/core/providers/runway"
"github.com/maximhq/bifrost/core/providers/sarvam"
"github.com/maximhq/bifrost/core/providers/sgl"
providerUtils "github.com/maximhq/bifrost/core/providers/utils"
"github.com/maximhq/bifrost/core/providers/vertex"
Expand Down Expand Up @@ -4291,6 +4292,8 @@ func (bifrost *Bifrost) createBaseProvider(providerKey schemas.ModelProvider, co
return runware.NewRunwareProvider(config, bifrost.logger)
case schemas.Fireworks:
return fireworks.NewFireworksProvider(config, bifrost.logger)
case schemas.Sarvam:
return sarvam.NewSarvamProvider(config, bifrost.logger)
default:
return nil, fmt.Errorf("unsupported provider: %s", targetProviderKey)
}
Expand Down
23 changes: 23 additions & 0 deletions core/internal/llmtests/account.go
Original file line number Diff line number Diff line change
Expand Up @@ -186,6 +186,7 @@ func (account *ComprehensiveTestAccount) GetConfiguredProviders() ([]schemas.Mod
schemas.Runway,
schemas.Runware,
schemas.Fireworks,
schemas.Sarvam,
ProviderOpenAICustom,
}, nil
}
Expand Down Expand Up @@ -564,6 +565,15 @@ func (account *ComprehensiveTestAccount) GetKeysForProvider(ctx context.Context,
},
},
}, nil
case schemas.Sarvam:
return []schemas.Key{
{
Value: *schemas.NewSecretVar("env.SARVAM_API_KEY"),
Models: []string{"*"},
Weight: 1.0,
UseForBatchAPI: new(true),
},
}, nil
default:
return nil, fmt.Errorf("unsupported provider: %s", providerKey)
}
Expand Down Expand Up @@ -940,6 +950,19 @@ func (account *ComprehensiveTestAccount) GetConfigForProvider(providerKey schema
BufferSize: 10,
},
}, nil
case schemas.Sarvam:
return &schemas.ProviderConfig{
NetworkConfig: schemas.NetworkConfig{
DefaultRequestTimeoutInSeconds: 120,
MaxRetries: 10,
RetryBackoffInitial: 1 * time.Second,
RetryBackoffMax: 12 * time.Second,
},
ConcurrencyAndBufferSize: schemas.ConcurrencyAndBufferSize{
Concurrency: Concurrency,
BufferSize: 10,
},
}, nil
default:
return nil, fmt.Errorf("unsupported provider: %s", providerKey)
}
Expand Down
33 changes: 26 additions & 7 deletions core/internal/llmtests/chat_completion_stream.go
Original file line number Diff line number Diff line change
Expand Up @@ -75,13 +75,15 @@ func RunChatCompletionStreamTest(t *testing.T, client *bifrost.Bifrost, ctx cont
CreateBasicChatMessage("Tell me a short story about a robot learning to paint the city which has the eiffel tower. Keep it under 200 words and include the city's name."),
}

params := &schemas.ChatParameters{
MaxCompletionTokens: bifrost.Ptr(1000),
}

request := &schemas.BifrostChatRequest{
Provider: testConfig.Provider,
Model: testConfig.ChatModel,
Input: messages,
Params: &schemas.ChatParameters{
MaxCompletionTokens: bifrost.Ptr(1000),
},
Provider: testConfig.Provider,
Model: testConfig.ChatModel,
Input: messages,
Params: params,
Fallbacks: testConfig.Fallbacks,
}

Expand Down Expand Up @@ -210,7 +212,24 @@ func RunChatCompletionStreamTest(t *testing.T, client *bifrost.Bifrost, ctx cont
responseCount++

// Safety check to prevent infinite loops in case of issues
if responseCount > 500 {
maxChunks := 500
if testConfig.Provider == schemas.Sarvam {
// Sarvam's own docs confirm reasoning is on by default and its
// tokens count toward the completion budget; documented workaround
// is "increase max_tokens, or disable reasoning with
// reasoning_effort=None". Verified live against sarvam-105b: a
// single response for this long-form prompt streamed ~1400 total
// chunks (reasoning_content + content). reasoning_effort:"low"
// barely reduces this, and the literal string "none" is rejected
// outright by Sarvam's API (400: "Input should be 'low', 'medium'
// or 'high'") - only a JSON null reliably disables it, which
// Bifrost's typed ChatParameters.Reasoning.Effort (*string,
// omitempty) can't emit. So raise the cap here instead of skipping
// the scenario - this still guards against genuine infinite loops,
// just with headroom for Sarvam's verbose default reasoning.
maxChunks = 3000
}
if responseCount > maxChunks {
t.Fatal("Received too many streaming chunks, something might be wrong")
}

Expand Down
47 changes: 33 additions & 14 deletions core/internal/llmtests/responses_stream.go
Original file line number Diff line number Diff line change
Expand Up @@ -33,13 +33,15 @@ func RunResponsesStreamTest(t *testing.T, client *bifrost.Bifrost, ctx context.C
},
}

responsesParams := &schemas.ResponsesParameters{
MaxOutputTokens: bifrost.Ptr(300),
}

request := &schemas.BifrostResponsesRequest{
Provider: testConfig.Provider,
Model: testConfig.ChatModel,
Input: messages,
Params: &schemas.ResponsesParameters{
MaxOutputTokens: bifrost.Ptr(300),
},
Provider: testConfig.Provider,
Model: testConfig.ChatModel,
Input: messages,
Params: responsesParams,
Fallbacks: testConfig.Fallbacks,
}

Expand Down Expand Up @@ -250,7 +252,14 @@ func RunResponsesStreamTest(t *testing.T, client *bifrost.Bifrost, ctx context.C
responseCount++

// Safety check to prevent infinite loops
if responseCount > 500 {
maxChunks := 500
if testConfig.Provider == schemas.Sarvam {
// Sarvam's default-on reasoning generates far more chunks than
// other providers for long-form prompts; see the matching
// comment in RunChatCompletionStreamTest (chat_completion_stream.go).
maxChunks = 3000
}
if responseCount > maxChunks {
return ResponsesStreamValidationResult{
Passed: false,
Errors: []string{"❌ Received too many streaming chunks, something might be wrong"},
Expand Down Expand Up @@ -684,13 +693,15 @@ func RunResponsesStreamTest(t *testing.T, client *bifrost.Bifrost, ctx context.C
},
}

lifecycleParams := &schemas.ResponsesParameters{
MaxOutputTokens: bifrost.Ptr(500),
}

request := &schemas.BifrostResponsesRequest{
Provider: testConfig.Provider,
Model: testConfig.ChatModel,
Input: messages,
Params: &schemas.ResponsesParameters{
MaxOutputTokens: bifrost.Ptr(500),
},
Provider: testConfig.Provider,
Model: testConfig.ChatModel,
Input: messages,
Params: lifecycleParams,
Fallbacks: testConfig.Fallbacks,
}

Expand Down Expand Up @@ -835,7 +846,15 @@ func RunResponsesStreamTest(t *testing.T, client *bifrost.Bifrost, ctx context.C
}

// Safety check to prevent infinite loops
if responseCount > 300 {
maxLifecycleChunks := 300
if testConfig.Provider == schemas.Sarvam {
// Sarvam's default-on reasoning generates far more chunks than
// other providers before terminal lifecycle events arrive; see
// the matching comment in RunChatCompletionStreamTest
// (chat_completion_stream.go).
maxLifecycleChunks = 3000
}
if responseCount > maxLifecycleChunks {
goto lifecycleComplete
}

Expand Down
52 changes: 52 additions & 0 deletions core/providers/sarvam/TODO.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
# Sarvam provider — deferred work

## Reasoning-token usage breakdown not reported

**Status:** not implemented, intentionally deferred pending a concrete need.

**What's missing:** `schemas.ChatCompletionTokensDetails.ReasoningTokens` (and `.ReasoningTokensCost`)
are always `0` for Sarvam responses. This is the same field the Anthropic
thinking-tokens usage fix (this repo, PR preceding this branch) populates
from `output_tokens_details.thinking_tokens`.

**Why:** Sarvam's `usage.completion_tokens_details` is always `null` in every
response observed (verified against both live responses and Sarvam's
published OpenAPI spec — `CompletionUsage.completion_tokens_details` is a
loosely-typed open object with no fixed schema, but Sarvam never populates
it). There is no other field carrying a reasoning-token count anywhere in
Sarvam's chat completion response — `ChatCompletionResponseMessage` pairs
`reasoning_content` (the raw text) with no accompanying count.

**What is correct today:** `usage.completion_tokens` / `usage.total_tokens`
already include reasoning-token consumption in the total (confirmed live: a
request that spent ~999 of 1000 completion tokens on reasoning reported
`completion_tokens: 1001`, matching the real total). Budget enforcement,
quota tracking, and cost totals that operate on the aggregate token count are
unaffected. What's missing is purely the *category breakdown* (reasoning vs.
output) for reporting/attribution — not the total itself.

**Considered and rejected (for now):** client-side estimation, i.e.
tokenizing `reasoning_content` ourselves to synthesize a `ReasoningTokens`
value Sarvam never sent. Rejected because:
- It would be an estimate against an unknown tokenizer (Sarvam doesn't
publish one), so the number could be meaningfully wrong — arguably worse
for a billing/governance field than reporting zero.
- No other provider without native `reasoning_tokens` gets this estimation
treatment in Bifrost today; adding it only for Sarvam would be a one-off
inconsistency, not a documented pattern.
- Out of the original scope (chat/TTS/STT wire compatibility) — this
surfaced from a governance question during review, not a filed request.
- Adds a new tokenizer dependency for a single provider's cosmetic field.

**When to revisit:** if there's a concrete downstream need for
per-category (thinking vs. output) cost attribution/reporting specifically
for Sarvam usage, and an approximate number is acceptable. If so:
1. Pick a tokenizer approximation (e.g. reuse whatever the repo already
vendors for other estimation needs, if any).
2. Populate `ChatCompletionTokensDetails.ReasoningTokens` from a token count
of `message.reasoning_content` in `core/providers/sarvam/sarvam.go`'s
chat path (would need a thin wrapper around
`openai.HandleOpenAIChatCompletionRequest`'s response, since that call is
currently a straight passthrough with no Sarvam-specific post-processing).
3. Clearly label the value as estimated (not authoritative) wherever it
surfaces, so it isn't confused with a real provider-reported count.
33 changes: 33 additions & 0 deletions core/providers/sarvam/cachedcontents.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
package sarvam

import (
providerUtils "github.com/maximhq/bifrost/core/providers/utils"
"github.com/maximhq/bifrost/core/schemas"
)

// CachedContentCreate is unsupported on SarvamProvider. Only Gemini and Vertex AI
// implement the cached-content lifecycle (Google AI Studio + Vertex AI named
// caches). Sarvam has no equivalent.
func (provider *SarvamProvider) CachedContentCreate(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCachedContentCreateRequest) (*schemas.BifrostCachedContentCreateResponse, *schemas.BifrostError) {
return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentCreateRequest, provider.GetProviderKey())
}

// CachedContentList is unsupported on SarvamProvider (see CachedContentCreate).
func (provider *SarvamProvider) CachedContentList(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentListRequest) (*schemas.BifrostCachedContentListResponse, *schemas.BifrostError) {
return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentListRequest, provider.GetProviderKey())
}

// CachedContentRetrieve is unsupported on SarvamProvider (see CachedContentCreate).
func (provider *SarvamProvider) CachedContentRetrieve(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentRetrieveRequest) (*schemas.BifrostCachedContentRetrieveResponse, *schemas.BifrostError) {
return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentRetrieveRequest, provider.GetProviderKey())
}

// CachedContentUpdate is unsupported on SarvamProvider (see CachedContentCreate).
func (provider *SarvamProvider) CachedContentUpdate(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentUpdateRequest) (*schemas.BifrostCachedContentUpdateResponse, *schemas.BifrostError) {
return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentUpdateRequest, provider.GetProviderKey())
}

// CachedContentDelete is unsupported on SarvamProvider (see CachedContentCreate).
func (provider *SarvamProvider) CachedContentDelete(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentDeleteRequest) (*schemas.BifrostCachedContentDeleteResponse, *schemas.BifrostError) {
return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentDeleteRequest, provider.GetProviderKey())
}
21 changes: 21 additions & 0 deletions core/providers/sarvam/errors.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
package sarvam

import (
providerUtils "github.com/maximhq/bifrost/core/providers/utils"
schemas "github.com/maximhq/bifrost/core/schemas"
"github.com/valyala/fasthttp"
)

// parseSarvamError parses Sarvam's error envelope: {"error":{"message","code","request_id"}}.
func parseSarvamError(resp *fasthttp.Response) *schemas.BifrostError {
var errorResp SarvamError
bifrostErr := providerUtils.HandleProviderAPIError(resp, &errorResp)
if errorResp.Error != nil {
if bifrostErr.Error == nil {
bifrostErr.Error = &schemas.ErrorField{}
}
bifrostErr.Error.Message = errorResp.Error.Message
bifrostErr.Error.Type = new(errorResp.Error.Code)
}
return bifrostErr
}
Loading