Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion core/changelog.md
Original file line number Diff line number Diff line change
@@ -1,4 +1,3 @@
- fix: forward OpenCode Responses requests directly to /v1/responses [@mohammadrezwankhan](https://github.com/mohammadrezwankhan)
- feat: add the `VideoEdit` operation with `BifrostVideoEditRequest`, `VideoEditInput` and `VideoEditParameters` for prompt-driven edits, upscaling and background removal on an existing video supplied as bytes, a URL or a provider video ID; implemented for OpenAI (`/v1/videos/edits`) and Runware (`videoInference`, `upscale`, `removeBackground`), with the model optional when the source is a video ID and the prompt optional for asset-driven task types (#6270)
- feat: batch accounting: `MergeBifrostLLMUsage` promoted to `schemas`, `Endpoint` on `BifrostBatchResultsResponse`, `BatchResultItem.Failed()`, `BatchRequestCountsFromResults`, `BatchRequestCounts.IsZero()`, raw-JSON Gemini batch result parsing and `custom_id` validation in `ConvertRequestsToJSONL`; the settlement engine (`AccountBatchResults` with runner-ID ownership fencing, idempotent aggregate log writes and governance reporting) and a sweeper that polls due jobs with capped, jittered backoff; aggregate log entries carry a `bifrost/<version>` user agent via `BifrostContextKeyRuntimeVersion` (thanks [@SahilChoudhary22](https://github.com/SahilChoudhary22)!) (#5291, #5294, #6474)
- feat: Claude-on-Vertex batch support: `ToVertexBatchCreateRequest` resolves Anthropic families to `publishers/anthropic/models/...`, `vertexConvertRequestsToJSONL` emits Claude-on-Vertex instances, `custom_id` round-trips through `batchResultsByKey`, and `GeminiBatchGenerateContentRequest` keeps `tools`, `toolConfig`, `cachedContent`, `labels` and the display name (#5368)
Expand All @@ -23,6 +22,7 @@
- feat: `ResponsesResponseError.Type` and a shared Responses stream-error normalizer so terminal `error`/`response.failed` events inside an HTTP 200 Azure SSE stream surface as errors with their nested type, code and message on both create-stream and retrieve-stream paths (thanks [@dani29](https://github.com/dani29)!) (#6302)
- feat: `ServiceTier` on `StreamAccumulatorResult`, with Anthropic's `service_tier` from `message_start` latched onto the final chunk of chat and Responses streams (#6236)
- feat: OpenRouter speech and transcription through the OpenAI-compatible audio handlers instead of returning unsupported-operation errors (#5734)
- fix: preserve xhigh reasoning effort for OpenAI-compatible destinations (vLLM, Ollama, SGL, and custom providers) instead of rewriting it to high using OpenAI model-name rules [@pranavthakur0-0](https://github.com/pranavthakur0-0) (#6193)
- fix: preserve `max_tokens` for OpenCode-compatible chat endpoints (thanks [@Alex-wangyang](https://github.com/Alex-wangyang)!) (#6458)
- fix: HuggingFace chat streaming completed with zero tokens and therefore zero cost while non-streaming calls on the same models priced correctly, for two reasons: HuggingFace was listed as a provider that omits the `[DONE]` marker (it sends one), which made the shared OpenAI streaming loop `break` on the first `finish_reason` and discard the trailing usage-only chunk that several router inference providers emit; and `stream_options.include_usage` never reached the router because the shared streaming handler returns early when a provider supplies a custom request converter. Both are corrected, and an explicit `stream_options` from the caller still wins (thanks [@elliottrabac](https://github.com/elliottrabac)!) (#6478)
- fix: preserve the caller's JSON Schema key order for structured outputs - `ChatParameters.UnmarshalJSON` holds `response_format` as raw bytes and the new `ChatResponseFormat` reader splices them verbatim into OpenAI, Anthropic, Bedrock, Gemini (unless a union `type` array needs normalizing) and Cohere requests, and `ResponsesTextConfigFormatJSONSchema` re-encodes in the decoded key sequence, because OpenAI structured outputs generate fields in the declared order and a re-sorted schema silently changes model behavior (#6235)
Expand Down
2 changes: 1 addition & 1 deletion core/providers/openai/chat.go
Original file line number Diff line number Diff line change
Expand Up @@ -222,7 +222,7 @@ func (req *OpenAIChatRequest) normalizeReasoningEffort(caps schemas.ModelCaps) {
if req.ChatParameters.Reasoning.Effort != nil {
// Native field is provided, use it (and clear max_tokens)
effort := *req.ChatParameters.Reasoning.Effort
req.ChatParameters.Reasoning.Effort = schemas.Ptr(caps.NormalizeReasoningEffort(effort, defaultEffortControl(caps.Model())))
req.ChatParameters.Reasoning.Effort = schemas.Ptr(normalizeReasoningEffort(req.Provider, caps, effort))
// Clear max_tokens since OpenAI doesn't use it
req.ChatParameters.Reasoning.MaxTokens = nil
} else if req.ChatParameters.Reasoning.MaxTokens != nil {
Expand Down
2 changes: 1 addition & 1 deletion core/providers/openai/responses.go
Original file line number Diff line number Diff line change
Expand Up @@ -322,7 +322,7 @@ func ToOpenAIResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.B
if req.ResponsesParameters.Reasoning.Effort != nil {
// Native field is provided, use it (and clear max_tokens)
effort := *req.ResponsesParameters.Reasoning.Effort
req.ResponsesParameters.Reasoning.Effort = schemas.Ptr(caps.NormalizeReasoningEffort(effort, defaultEffortControl(capModel)))
req.ResponsesParameters.Reasoning.Effort = schemas.Ptr(normalizeReasoningEffort(req.Provider, caps, effort))
// Clear max_tokens since OpenAI doesn't use it
req.ResponsesParameters.Reasoning.MaxTokens = nil
} else if req.ResponsesParameters.Reasoning.MaxTokens != nil {
Expand Down
76 changes: 51 additions & 25 deletions core/providers/openai/responses_marshal_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -159,39 +159,65 @@ func TestOpenAIResponsesRequest_MarshalJSON_ReasoningMaxTokensAbsent(t *testing.
func TestNormalizeOpenAIReasoningEffort(t *testing.T) {
tests := []struct {
name string
provider schemas.ModelProvider
model string
effort string
expected string
}{
{"preserves minimal for gpt-5", "gpt-5", "minimal", "minimal"},
{"preserves minimal for gpt-5-mini", "gpt-5-mini", "minimal", "minimal"},
{"preserves minimal for gpt-5-nano", "gpt-5-nano", "minimal", "minimal"},
{"maps minimal to low for gpt-5.4", "gpt-5.4", "minimal", "low"},
{"maps minimal to low for gpt-5.2", "gpt-5.2", "minimal", "low"},
{"maps minimal to low for gpt-5.6", "gpt-5.6", "minimal", "low"},
{"maps minimal to low for o3", "o3", "minimal", "low"},
{"maps minimal to low for o1", "o1", "minimal", "low"},
{"maps minimal to low for o4", "o4", "minimal", "low"},
{"maps minimal to low for gpt-oss", "gpt-oss", "minimal", "low"},
{"gpt-5.6 keeps max", "gpt-5.6", "max", "max"},
{"gpt-5.6 variant keeps max", "gpt-5.6-terra", "max", "max"},
{"gpt-5.6 keeps xhigh", "gpt-5.6", "xhigh", "xhigh"},
{"provider-prefixed gpt-5.6 keeps max", "openai/gpt-5.6", "max", "max"},
{"deepseek-v4 keeps max", "deepseek-v4", "max", "max"},
{"glm-5.2 keeps max", "glm-5.2", "max", "max"},
{"gpt-5.5 downgrades max to xhigh", "gpt-5.5", "max", "xhigh"},
{"gpt-5.2 downgrades max to xhigh", "gpt-5.2", "max", "xhigh"},
{"gpt-5.5 keeps xhigh", "gpt-5.5", "xhigh", "xhigh"},
{"gpt-5.1 downgrades max to high", "gpt-5.1", "max", "high"},
{"gpt-5.1 downgrades xhigh to high", "gpt-5.1", "xhigh", "high"},
{"standard effort passes through", "gpt-5.1", "medium", "medium"},
{"preserves minimal for gpt-5", schemas.OpenAI, "gpt-5", "minimal", "minimal"},
{"preserves minimal for gpt-5-mini", schemas.OpenAI, "gpt-5-mini", "minimal", "minimal"},
{"preserves minimal for gpt-5-nano", schemas.OpenAI, "gpt-5-nano", "minimal", "minimal"},
{"maps minimal to low for gpt-5.4", schemas.OpenAI, "gpt-5.4", "minimal", "low"},
{"maps minimal to low for gpt-5.2", schemas.OpenAI, "gpt-5.2", "minimal", "low"},
{"maps minimal to low for gpt-5.6", schemas.OpenAI, "gpt-5.6", "minimal", "low"},
{"maps minimal to low for o3", schemas.OpenAI, "o3", "minimal", "low"},
{"maps minimal to low for o1", schemas.OpenAI, "o1", "minimal", "low"},
{"maps minimal to low for o4", schemas.OpenAI, "o4", "minimal", "low"},
{"maps minimal to low for gpt-oss", schemas.OpenAI, "gpt-oss", "minimal", "low"},
{"gpt-5.6 keeps max", schemas.OpenAI, "gpt-5.6", "max", "max"},
{"gpt-5.6 variant keeps max", schemas.OpenAI, "gpt-5.6-terra", "max", "max"},
{"gpt-5.6 keeps xhigh", schemas.OpenAI, "gpt-5.6", "xhigh", "xhigh"},
{"provider-prefixed gpt-5.6 keeps max", schemas.OpenAI, "openai/gpt-5.6", "max", "max"},
{"deepseek-v4 keeps max", schemas.OpenAI, "deepseek-v4", "max", "max"},
{"glm-5.2 keeps max", schemas.OpenAI, "glm-5.2", "max", "max"},
{"gpt-5.5 downgrades max to xhigh", schemas.OpenAI, "gpt-5.5", "max", "xhigh"},
{"gpt-5.2 downgrades max to xhigh", schemas.OpenAI, "gpt-5.2", "max", "xhigh"},
{"gpt-5.5 keeps xhigh", schemas.OpenAI, "gpt-5.5", "xhigh", "xhigh"},
{"gpt-5.1 downgrades max to high", schemas.OpenAI, "gpt-5.1", "max", "high"},
{"gpt-5.1 downgrades xhigh to high", schemas.OpenAI, "gpt-5.1", "xhigh", "high"},
{"standard effort passes through", schemas.OpenAI, "gpt-5.1", "medium", "medium"},
{"vllm preserves xhigh for qwen", schemas.VLLM, "qwen/qwen3.8-27b", "xhigh", "xhigh"},
{"vllm maps max to xhigh for qwen", schemas.VLLM, "qwen/qwen3.8-27b", "max", "xhigh"},
{"vllm still maps minimal to low", schemas.VLLM, "qwen/qwen3.8-27b", "minimal", "low"},
{"ollama preserves xhigh", schemas.Ollama, "qwen3.8:27b", "xhigh", "xhigh"},
{"sgl preserves xhigh", schemas.SGL, "qwen/qwen3.8-27b", "xhigh", "xhigh"},
{"custom openai-compatible provider preserves xhigh", schemas.ModelProvider("gpustack-rtxpro-6000"), "qwen/qwen3.8-27b", "xhigh", "xhigh"},
{"openai still downgrades xhigh on qwen-named models", schemas.OpenAI, "qwen/qwen3.8-27b", "xhigh", "high"},
}

for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
caps := schemas.ResolveModelCaps(schemas.OpenAI, tt.model)
if got := caps.NormalizeReasoningEffort(tt.effort, defaultEffortControl(tt.model)); got != tt.expected {
t.Errorf("model %q: NormalizeReasoningEffort(%q) = %q, want %q", tt.model, tt.effort, got, tt.expected)
caps := schemas.ResolveModelCaps(tt.provider, tt.model)
if got := caps.NormalizeReasoningEffort(tt.effort, defaultEffortControl(tt.provider, tt.model)); got != tt.expected {
t.Errorf("provider %q model %q: NormalizeReasoningEffort(%q) = %q, want %q", tt.provider, tt.model, tt.effort, got, tt.expected)
}
})
}
}

func TestNormalizeReasoningEffortPreservesXHighForOpenAICompatibleProviders(t *testing.T) {
tests := []schemas.ModelProvider{
schemas.VLLM,
schemas.Ollama,
schemas.SGL,
schemas.ModelProvider("gpustack-rtxpro-6000"),
}

for _, provider := range tests {
t.Run(string(provider), func(t *testing.T) {
caps := schemas.ResolveModelCaps(provider, "qwen/qwen3.8-27b")
if got := normalizeReasoningEffort(provider, caps, schemas.ReasoningEffortXHigh); got != schemas.ReasoningEffortXHigh {
t.Fatalf("normalizeReasoningEffort(%q, xhigh) = %q, want xhigh", provider, got)
}
})
}
Expand Down
34 changes: 31 additions & 3 deletions core/providers/openai/utils.go
Original file line number Diff line number Diff line change
Expand Up @@ -55,12 +55,12 @@ func IsOpenAIReasoningModel(model string) bool {
// defaultEffortControl widens the base low/medium/high ladder with the effort
// levels a model natively accepts. Only the widening is name-derived; the
// datasheet's per-level booleans take precedence when a row exists.
func defaultEffortControl(model string) *schemas.EffortControl {
func defaultEffortControl(provider schemas.ModelProvider, model string) *schemas.EffortControl {
levels := []string{schemas.ReasoningEffortLow, schemas.ReasoningEffortMedium, schemas.ReasoningEffortHigh}
if acceptsMinimalEffort(model) {
levels = append([]string{schemas.ReasoningEffortMinimal}, levels...)
}
if acceptsXHighEffort(model) {
if includeXHighInDefaultLadder(provider, model) {
levels = append(levels, schemas.ReasoningEffortXHigh)
}
if acceptsMaxEffort(model) {
Expand All @@ -69,6 +69,35 @@ func defaultEffortControl(model string) *schemas.EffortControl {
return &schemas.EffortControl{Levels: levels}
}

// includeXHighInDefaultLadder reports whether the OpenAI-dialect fallback
// ladder should list "xhigh". OpenAI, Azure, and xAI still use the hosted
// catalog (gpt-5.2+ / Grok prefixes). Every other destination — built-in
// vLLM/Ollama/SGL and custom providers with base_provider_type openai —
// keeps the client value, because OpenAI model prefixes are the wrong
// oracle for those upstreams.
func includeXHighInDefaultLadder(provider schemas.ModelProvider, model string) bool {
switch provider {
case schemas.OpenAI, schemas.Azure, schemas.XAI:
return acceptsXHighEffort(model)
default:
return true
}
}

// normalizeReasoningEffort applies hosted-provider compatibility rules while
// preserving xhigh for OpenAI-compatible destinations. Their model names and
// capability records do not describe OpenAI's hosted effort enum.
func normalizeReasoningEffort(provider schemas.ModelProvider, caps schemas.ModelCaps, effort string) string {
if provider == "" {
provider = caps.Provider()
}
if effort == schemas.ReasoningEffortXHigh && provider != schemas.OpenAI &&
provider != schemas.Azure && provider != schemas.XAI {
return effort
}
return caps.NormalizeReasoningEffort(effort, defaultEffortControl(provider, caps.Model()))
}

// acceptsXHighEffort reports models that natively accept "xhigh" effort. The
// ladder is shared by every OpenAI-dialect provider, so non-OpenAI families
// supporting the tier are recognised here too — otherwise their "xhigh" is
Expand Down Expand Up @@ -126,7 +155,6 @@ func bareModelLower(model string) string {
return strings.ToLower(model)
}


func ConvertOpenAIMessagesToBifrostMessages(messages []OpenAIMessage) []schemas.ChatMessage {
bifrostMessages := make([]schemas.ChatMessage, len(messages))
for i, message := range messages {
Expand Down
185 changes: 185 additions & 0 deletions core/providers/vllm/chat_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -108,3 +108,188 @@ func keys(m map[string]interface{}) []string {
}
return out
}

func TestChatCompletion_XHighReasoningEffortPreserved(t *testing.T) {
t.Parallel()

var capturedBody map[string]interface{}

server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
if err != nil {
http.Error(w, "read error", http.StatusInternalServerError)
return
}
if err := json.Unmarshal(body, &capturedBody); err != nil {
http.Error(w, "json error", http.StatusInternalServerError)
return
}
w.Header().Set("Content-Type", "application/json")
if _, err := fmt.Fprint(w, `{
"id": "chatcmpl-test",
"object": "chat.completion",
"created": 1234567890,
"model": "qwen/qwen3.8-27b",
"choices": [{
"index": 0,
"message": {"role": "assistant", "content": "Hello!"},
"finish_reason": "stop"
}],
"usage": {"prompt_tokens": 5, "completion_tokens": 3, "total_tokens": 8}
}`); err != nil {
t.Errorf("write mock response: %v", err)
}
}))
defer server.Close()

provider := newTestVLLMProvider()
key := schemas.Key{
ID: "test-key",
Value: schemas.SecretVar{Val: "test-api-key"},
VLLMKeyConfig: &schemas.VLLMKeyConfig{
URL: schemas.SecretVar{Val: server.URL},
},
}

ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline)

hello := "Hello"
xhigh := "xhigh"
req := &schemas.BifrostChatRequest{
Provider: schemas.VLLM,
Model: "qwen/qwen3.8-27b",
Input: []schemas.ChatMessage{
{
Role: schemas.ChatMessageRoleUser,
Content: &schemas.ChatMessageContent{ContentStr: &hello},
},
},
Params: &schemas.ChatParameters{
Reasoning: &schemas.ChatReasoning{
Effort: &xhigh,
},
},
}

_, bifrostErr := provider.ChatCompletion(ctx, key, req)
if bifrostErr != nil {
t.Fatalf("ChatCompletion returned error: %v", bifrostErr.Error.Message)
}

if capturedBody == nil {
t.Fatal("mock server did not receive a request body")
}

rawEffort, ok := capturedBody["reasoning_effort"]
if !ok {
t.Fatalf("reasoning_effort missing from outgoing request body; got keys: %v", keys(capturedBody))
}

effortStr, ok := rawEffort.(string)
if !ok {
t.Fatalf("expected reasoning_effort to be a string, got %T", rawEffort)
}

if effortStr != "xhigh" {
t.Fatalf("expected reasoning_effort=xhigh, got %q", effortStr)
}
}

func TestResponses_XHighReasoningEffortPreserved(t *testing.T) {
t.Parallel()

var capturedBody map[string]interface{}

server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
if err != nil {
http.Error(w, "read error", http.StatusInternalServerError)
return
}
if err := json.Unmarshal(body, &capturedBody); err != nil {
http.Error(w, "json error", http.StatusInternalServerError)
return
}
w.Header().Set("Content-Type", "application/json")
if _, err := fmt.Fprint(w, `{
"id": "resp-test",
"object": "response",
"created_at": 1234567890,
"model": "qwen/qwen3.8-27b",
"status": "completed",
"output": [{
"type": "message",
"id": "msg_01",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}]
}],
"usage": {"input_tokens": 5, "output_tokens": 3, "total_tokens": 8}
}`); err != nil {
t.Errorf("write mock response: %v", err)
}
}))
defer server.Close()

provider := newTestVLLMProvider()
key := schemas.Key{
ID: "test-key",
Value: schemas.SecretVar{Val: "test-api-key"},
VLLMKeyConfig: &schemas.VLLMKeyConfig{
URL: schemas.SecretVar{Val: server.URL},
},
}

ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline)

hello := "Hello"
xhigh := "xhigh"
req := &schemas.BifrostResponsesRequest{
Provider: schemas.VLLM,
Model: "qwen/qwen3.8-27b",
Input: []schemas.ResponsesMessage{
{
Role: schemas.Ptr(schemas.ResponsesInputMessageRoleUser),
Content: &schemas.ResponsesMessageContent{ContentStr: &hello},
},
},
Params: &schemas.ResponsesParameters{
Reasoning: &schemas.ResponsesParametersReasoning{
Effort: &xhigh,
},
},
}

_, bifrostErr := provider.Responses(ctx, key, req)
if bifrostErr != nil {
t.Fatalf("Responses returned error: %v", bifrostErr.Error.Message)
}

if capturedBody == nil {
t.Fatal("mock server did not receive a request body")
}

// Responses API format uses reasoning: { effort: "xhigh" }
rawReasoning, ok := capturedBody["reasoning"]
if !ok {
t.Fatalf("reasoning missing from outgoing request body; got keys: %v", keys(capturedBody))
}

reasoningMap, ok := rawReasoning.(map[string]interface{})
if !ok {
t.Fatalf("expected reasoning to be an object, got %T", rawReasoning)
}

rawEffort, ok := reasoningMap["effort"]
if !ok {
t.Fatalf("effort missing from reasoning object; got keys: %v", keys(reasoningMap))
}

effortStr, ok := rawEffort.(string)
if !ok {
t.Fatalf("expected effort to be a string, got %T", rawEffort)
}

if effortStr != "xhigh" {
t.Fatalf("expected reasoning.effort=xhigh, got %q", effortStr)
}
}
Loading