diff --git a/Makefile b/Makefile
index 6bc34ee7140..217d946ed5c 100644
--- a/Makefile
+++ b/Makefile
@@ -468,6 +468,12 @@ _build-with-docker: # Internal target for Docker-based cross-compilation
exit 1; \
fi
+# NOTE: transports/Dockerfile sets GOWORK=off and resolves the framework module
+# from the published tag (go.mod), so plain `make docker-image` FAILS on branches
+# whose local framework/ is ahead of the latest release (e.g. missing packages
+# like framework/batchaccounting). Use `LOCAL=1 make docker-image` (Dockerfile.local,
+# go-workspace build) on such branches — a stale-layer image from the plain target
+# has already caused one silent-regression incident (2026-08-20).
docker-image: build-ui ## Build Docker image (LOCAL=1 to use Dockerfile.local)
@$(ECHO) "$(GREEN)Building Docker image...$(NC)"
$(eval GIT_SHA=$(shell git rev-parse --short HEAD))
diff --git a/core/bifrost.go b/core/bifrost.go
index 6d51dc5be55..799b7b766da 100644
--- a/core/bifrost.go
+++ b/core/bifrost.go
@@ -21,6 +21,7 @@ import (
"github.com/maximhq/bifrost/core/mcp"
"github.com/maximhq/bifrost/core/mcp/codemode/starlark"
"github.com/maximhq/bifrost/core/mcp/credstore"
+ "github.com/maximhq/bifrost/core/providers/alibaba"
"github.com/maximhq/bifrost/core/providers/anthropic"
"github.com/maximhq/bifrost/core/providers/azure"
"github.com/maximhq/bifrost/core/providers/bedrock"
@@ -33,6 +34,7 @@ import (
"github.com/maximhq/bifrost/core/providers/gemini"
"github.com/maximhq/bifrost/core/providers/groq"
"github.com/maximhq/bifrost/core/providers/huggingface"
+ "github.com/maximhq/bifrost/core/providers/kimi"
"github.com/maximhq/bifrost/core/providers/mistral"
"github.com/maximhq/bifrost/core/providers/nebius"
"github.com/maximhq/bifrost/core/providers/ollama"
@@ -51,6 +53,7 @@ import (
"github.com/maximhq/bifrost/core/providers/vllm"
"github.com/maximhq/bifrost/core/providers/wafer"
"github.com/maximhq/bifrost/core/providers/xai"
+ "github.com/maximhq/bifrost/core/providers/zhipu"
schemas "github.com/maximhq/bifrost/core/schemas"
"github.com/valyala/fasthttp"
)
@@ -4512,6 +4515,12 @@ func (bifrost *Bifrost) createBaseProvider(providerKey schemas.ModelProvider, co
return deepseek.NewDeepSeekProvider(config, bifrost.logger)
case schemas.Wafer:
return wafer.NewWaferProvider(config, bifrost.logger)
+ case schemas.Alibaba:
+ return alibaba.NewAlibabaProvider(config, bifrost.logger)
+ case schemas.Kimi:
+ return kimi.NewKimiProvider(config, bifrost.logger)
+ case schemas.Zhipu:
+ return zhipu.NewZhipuProvider(config, bifrost.logger)
case schemas.Gemini:
return gemini.NewGeminiProvider(config, bifrost.logger), nil
case schemas.OpenRouter:
diff --git a/core/changelog.md b/core/changelog.md
index 50c337596b4..a561acf1389 100644
--- a/core/changelog.md
+++ b/core/changelog.md
@@ -16,3 +16,5 @@
- fix: accept a bare model identifier on Bedrock rerank by synthesizing the foundation-model ARN from the resolved region - Rerank is the one Bedrock surface that names its model by ARN rather than by bare ID, so all three rerank drop-ins in the provider harness 400'd on `amazon.rerank-v1:0`. The partition is derived from the region (`aws`, `aws-cn`, `aws-us-gov`) so GovCloud and China build a correct ARN, and an explicit ARN still passes through untouched
- fix: stop stripping `file_url` from OpenAI-shaped chat file blocks on marshal - dropping it produced `{"type":"file","file":{}}` and an upstream complaint about a missing `file_id`, which hid the fact that a source had been discarded. Providers that cannot take a URL now say so by name, and any OpenAI-compatible endpoint that does accept one keeps working without a Bifrost change
- fix: leave URL content sources Bifrost cannot download in place on the OpenAI and native-Anthropic paths instead of failing the request - only `http(s)` is fetched, and whether a `gs://`, `s3://` or scheme-less reference is usable is the provider's call, so the source now travels as `{"type":"url"}` and the platform answers for itself
+- fix: support GLM-5.2+ for max reasoning effort and clamp GLM-5.3+ values to max/high/low on every OpenAI-dialect mount (medium→high, none/minimal→low, xhigh→max), mirroring Z.ai's Coding Plan coercion — the GLM-5.3 API errors on every other value [@is911](https://github.com/is911)
+- feat: add alibaba, kimi and zhipu providers — Alibaba Cloud Model Studio (Qwen/DashScope), Kimi (Moonshot AI) and Zhipu AI (GLM/Z.AI) — with default OpenAI-compatible mounts and optional Anthropic-compatible routing (per-key `use_anthropic_endpoints` through the shared Anthropic converters), per-vendor `reasoning_effort` shaping, and native Responses/embeddings where the vendor offers them. qwen3.8-max publishes `xhigh` as its top tier on the OpenAI-compatible mount (vendor enum: none/minimal/low/medium/high/xhigh), so the gateway forwards `xhigh` verbatim and clamps `max`→`xhigh` instead of forwarding a value the vendor 400'd until ~2026-08-22. On the Anthropic-compatible mounts the gateway now clamps `output_config.effort` per model family instead of the blanket `max`→`xhigh` for alibaba (the mount's own API page documents no effort field; live-verified it validates the value against its per-model chat-completions enum instead of mapping it server-side — per-model matrix from the Model Studio model-page docs supplied 2026-08-23: qwen3.8-max keeps the `max`→`xhigh` clamp; glm-5.3+ takes `max`/`high`/`low` with `xhigh`→`max`, `medium`→`high`, `minimal`/`none`→`low`; glm-5.2/5.1/5 and non-dated deepseek-v4-pro/flash take `high`/`max` with `xhigh`→`max`, `low`/`medium`/`minimal`/`none`→`high` (`low` is out of this family's enum, so the mildest tiers collapse onto the mildest valid value); the dated snapshots deepseek-v4-pro-0813/deepseek-v4-flash-0731 take `max`/`high`/`low` with `xhigh`/`medium`→`high`, `minimal`/`none`→`low`), sends the effort alone without a synthesized `thinking` field when one is set (Model Studio rejects `reasoning_effort` and `thinking_budget` together and engages thinking itself), and derives the mount base URL idempotently for all three vendors so a `base_url` already pointing at the Anthropic mount is used as-is instead of doubling the suffix — the path-rewriting suffix rules only apply on each vendor's own hosts (`*.aliyuncs.com`, `api.z.ai`/`open.bigmodel.cn`, and Kimi's kimi/moonshot hosts), so a custom or proxied base URL keeps its configured path and only gets the mount suffix appended [@is911](https://github.com/is911)
diff --git a/core/internal/llmtests/account.go b/core/internal/llmtests/account.go
index 1a975eac566..cb6b6cd5b39 100644
--- a/core/internal/llmtests/account.go
+++ b/core/internal/llmtests/account.go
@@ -194,6 +194,9 @@ func (account *ComprehensiveTestAccount) GetConfiguredProviders() ([]schemas.Mod
schemas.Fireworks,
schemas.Sarvam,
schemas.Wafer,
+ schemas.Alibaba,
+ schemas.Kimi,
+ schemas.Zhipu,
ProviderOpenAICustom,
}, nil
}
@@ -485,6 +488,33 @@ func (account *ComprehensiveTestAccount) GetKeysForProvider(ctx context.Context,
UseForBatchAPI: bifrost.Ptr(true),
},
}, nil
+ case schemas.Alibaba:
+ return []schemas.Key{
+ {
+ Value: *schemas.NewSecretVar("env.ALIBABA_API_KEY"),
+ Models: []string{"*"},
+ Weight: 1.0,
+ UseForBatchAPI: bifrost.Ptr(true),
+ },
+ }, nil
+ case schemas.Kimi:
+ return []schemas.Key{
+ {
+ Value: *schemas.NewSecretVar("env.KIMI_API_KEY"),
+ Models: []string{"*"},
+ Weight: 1.0,
+ UseForBatchAPI: bifrost.Ptr(true),
+ },
+ }, nil
+ case schemas.Zhipu:
+ return []schemas.Key{
+ {
+ Value: *schemas.NewSecretVar("env.ZHIPU_API_KEY"),
+ Models: []string{"*"},
+ Weight: 1.0,
+ UseForBatchAPI: bifrost.Ptr(true),
+ },
+ }, nil
case schemas.Wafer:
return []schemas.Key{
{
@@ -872,6 +902,46 @@ func (account *ComprehensiveTestAccount) GetConfigForProvider(providerKey schema
BufferSize: 10,
},
}, nil
+ case schemas.Alibaba:
+ return &schemas.ProviderConfig{
+ NetworkConfig: schemas.NetworkConfig{
+ // Thinking models on 1M-context hosts need a generous timeout.
+ DefaultRequestTimeoutInSeconds: 180,
+ MaxRetries: 10,
+ RetryBackoffInitial: 5 * time.Second,
+ RetryBackoffMax: 3 * time.Minute,
+ },
+ ConcurrencyAndBufferSize: schemas.ConcurrencyAndBufferSize{
+ Concurrency: Concurrency,
+ BufferSize: 10,
+ },
+ }, nil
+ case schemas.Kimi:
+ return &schemas.ProviderConfig{
+ NetworkConfig: schemas.NetworkConfig{
+ DefaultRequestTimeoutInSeconds: 180,
+ MaxRetries: 10,
+ RetryBackoffInitial: 5 * time.Second,
+ RetryBackoffMax: 3 * time.Minute,
+ },
+ ConcurrencyAndBufferSize: schemas.ConcurrencyAndBufferSize{
+ Concurrency: Concurrency,
+ BufferSize: 10,
+ },
+ }, nil
+ case schemas.Zhipu:
+ return &schemas.ProviderConfig{
+ NetworkConfig: schemas.NetworkConfig{
+ DefaultRequestTimeoutInSeconds: 180,
+ MaxRetries: 10,
+ RetryBackoffInitial: 5 * time.Second,
+ RetryBackoffMax: 3 * time.Minute,
+ },
+ ConcurrencyAndBufferSize: schemas.ConcurrencyAndBufferSize{
+ Concurrency: Concurrency,
+ BufferSize: 10,
+ },
+ }, nil
case schemas.Wafer:
return &schemas.ProviderConfig{
NetworkConfig: schemas.NetworkConfig{
diff --git a/core/internal/llmtests/chat_completion_stream.go b/core/internal/llmtests/chat_completion_stream.go
index 2a1afc6a22d..46c3a406fe9 100644
--- a/core/internal/llmtests/chat_completion_stream.go
+++ b/core/internal/llmtests/chat_completion_stream.go
@@ -209,8 +209,10 @@ func RunChatCompletionStreamTest(t *testing.T, client *bifrost.Bifrost, ctx cont
responseCount++
- // Safety check to prevent infinite loops in case of issues
- if responseCount > 500 {
+ // Safety check to prevent infinite loops in case of issues.
+ // Generous bound: token-by-token reasoning streams (e.g. GLM) can
+ // legitimately produce well over 500 chunks for a single response.
+ if responseCount > 2500 {
t.Fatal("Received too many streaming chunks, something might be wrong")
}
@@ -406,7 +408,11 @@ func RunChatCompletionStreamTest(t *testing.T, client *bifrost.Bifrost, ctx cont
}
}
- if responseCount > 100 {
+ if responseCount > 1500 {
+ // Runaway stream: the tool-detection window is a safety bound,
+ // so tripping it must fail validation instead of letting a
+ // pathological stream pass on the strength of an earlier tool event.
+ streamErrors = append(streamErrors, "❌ Received too many streaming chunks in tool-call stream, something might be wrong")
goto toolStreamComplete
}
diff --git a/core/internal/llmtests/responses_stream.go b/core/internal/llmtests/responses_stream.go
index d21c3ba21fa..2d1602d08a6 100644
--- a/core/internal/llmtests/responses_stream.go
+++ b/core/internal/llmtests/responses_stream.go
@@ -249,8 +249,10 @@ func RunResponsesStreamTest(t *testing.T, client *bifrost.Bifrost, ctx context.C
responseCount++
- // Safety check to prevent infinite loops
- if responseCount > 500 {
+ // Safety check to prevent infinite loops.
+ // Generous bound: token-by-token reasoning streams (e.g. GLM)
+ // can legitimately produce well over 500 chunks.
+ if responseCount > 2500 {
return ResponsesStreamValidationResult{
Passed: false,
Errors: []string{"❌ Received too many streaming chunks, something might be wrong"},
@@ -484,8 +486,10 @@ func RunResponsesStreamTest(t *testing.T, client *bifrost.Bifrost, ctx context.C
}
}
- if responseCount > 100 {
- goto toolStreamComplete
+ if responseCount > 1500 {
+ // Runaway stream: tripping the tool-detection safety bound must
+ // fail the test — an earlier tool event cannot mask it.
+ t.Fatalf("❌ Received too many streaming chunks in tool-call stream (%d), something might be wrong", responseCount)
}
case <-streamCtx.Done():
diff --git a/core/internal/llmtests/validation_presets.go b/core/internal/llmtests/validation_presets.go
index 8d14d5665a9..28a96eb5144 100644
--- a/core/internal/llmtests/validation_presets.go
+++ b/core/internal/llmtests/validation_presets.go
@@ -486,6 +486,18 @@ func ModifyExpectationsForProvider(expectations ResponseExpectations, provider s
expectations.ShouldHaveUsageStats = true
expectations.ShouldHaveLatency = true
+ case schemas.Alibaba:
+ expectations.ShouldHaveUsageStats = true
+ expectations.ShouldHaveLatency = true
+
+ case schemas.Kimi:
+ expectations.ShouldHaveUsageStats = true
+ expectations.ShouldHaveLatency = true
+
+ case schemas.Zhipu:
+ expectations.ShouldHaveUsageStats = true
+ expectations.ShouldHaveLatency = true
+
case schemas.Wafer:
expectations.ShouldHaveUsageStats = true
expectations.ShouldHaveLatency = true
diff --git a/core/providers/alibaba/alibaba.go b/core/providers/alibaba/alibaba.go
new file mode 100644
index 00000000000..239930e2327
--- /dev/null
+++ b/core/providers/alibaba/alibaba.go
@@ -0,0 +1,550 @@
+// Package alibaba implements the Alibaba Cloud Model Studio (Qwen / Bailian / DashScope)
+// LLM provider.
+//
+// Model Studio serves OpenAI-compatible, Anthropic-compatible, and DashScope-native
+// mounts across several host shapes (legacy shared, workspace-dedicated, trial,
+// Token Plan, Coding Plan); each billing mode uses its own keys and rejects mismatched
+// key/base-URL pairs (doc: docs/research 02-provider-alibaba-cloud.md). The provider
+// defaults to the pay-as-you-go international host; subscription plans are reached by
+// overriding base_url. The Anthropic-compatible mount is selected per key/alias via
+// use_anthropic_endpoints.
+package alibaba
+
+import (
+ "context"
+ "maps"
+ "strings"
+ "time"
+
+ "github.com/maximhq/bifrost/core/providers/anthropic"
+ "github.com/maximhq/bifrost/core/providers/openai"
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ schemas "github.com/maximhq/bifrost/core/schemas"
+ "github.com/valyala/fasthttp"
+)
+
+// AlibabaProvider implements the Provider interface for Alibaba Cloud Model Studio.
+type AlibabaProvider struct {
+ logger schemas.Logger // Logger for provider operations
+ client *fasthttp.Client // HTTP client for unary API requests (ReadTimeout bounds overall response)
+ streamingClient *fasthttp.Client // HTTP client for streaming API requests (no ReadTimeout; idle governed by NewIdleTimeoutReader)
+ networkConfig schemas.NetworkConfig // Network configuration including extra headers
+ sendBackRawRequest bool // Whether to include raw request in BifrostResponse
+ sendBackRawResponse bool // Whether to include raw response in BifrostResponse
+}
+
+// NewAlibabaProvider creates a new Alibaba provider instance.
+// It initializes the HTTP client with the provided configuration and sets up response pools.
+// The client is configured with timeouts, concurrency limits, and optional proxy settings.
+func NewAlibabaProvider(config *schemas.ProviderConfig, logger schemas.Logger) (*AlibabaProvider, error) {
+ config.CheckAndSetDefaults()
+
+ // Clone the NetworkConfig (including its mutable maps) so the provider never
+ // shares state with the caller's ProviderConfig — later mutations of the
+ // caller's ExtraHeaders/BetaHeaderOverrides must not affect live requests,
+ // and the BaseURL defaulting below must not write back to the caller.
+ networkConfig := config.NetworkConfig
+ networkConfig.ExtraHeaders = maps.Clone(config.NetworkConfig.ExtraHeaders)
+ networkConfig.BetaHeaderOverrides = maps.Clone(config.NetworkConfig.BetaHeaderOverrides)
+
+ requestTimeout := time.Second * time.Duration(networkConfig.DefaultRequestTimeoutInSeconds)
+ client := &fasthttp.Client{
+ ReadTimeout: requestTimeout,
+ WriteTimeout: requestTimeout,
+ MaxConnsPerHost: networkConfig.MaxConnsPerHost,
+ MaxIdleConnDuration: time.Second * time.Duration(networkConfig.KeepAliveTimeoutInSeconds),
+ MaxConnWaitTimeout: requestTimeout,
+ MaxConnDuration: time.Second * time.Duration(schemas.DefaultMaxConnDurationInSeconds),
+ ConnPoolStrategy: fasthttp.FIFO,
+ }
+
+ // Configure proxy and retry policy
+ client = providerUtils.ConfigureProxy(client, config.ProxyConfig, logger)
+ client = providerUtils.ConfigureDialer(client, networkConfig.AllowPrivateNetwork)
+ client = providerUtils.ConfigureTLS(client, networkConfig, logger)
+ streamingClient := providerUtils.BuildStreamingClient(client)
+ // Set default BaseURL if not provided
+ if networkConfig.BaseURL == "" {
+ networkConfig.BaseURL = defaultBaseURL
+ }
+ networkConfig.BaseURL = strings.TrimRight(networkConfig.BaseURL, "/")
+
+ return &AlibabaProvider{
+ logger: logger,
+ client: client,
+ streamingClient: streamingClient,
+ networkConfig: networkConfig,
+ sendBackRawRequest: config.SendBackRawRequest,
+ sendBackRawResponse: config.SendBackRawResponse,
+ }, nil
+}
+
+// xAPIKeyHeaders builds the auth headers for Alibaba's Anthropic-compatible mount
+// (accepts x-api-key; Authorization: Bearer also works upstream).
+func (provider *AlibabaProvider) xAPIKeyHeaders(key schemas.Key) map[string]string {
+ headers := map[string]string{}
+ if key.Value.GetValue() != "" {
+ headers["x-api-key"] = key.Value.GetValue()
+ }
+ return headers
+}
+
+// anthropicMessagesURL returns the full Anthropic-mount messages URL for this request,
+// derived from the configured OpenAI base URL and honoring per-request path overrides.
+func (provider *AlibabaProvider) anthropicMessagesURL(ctx *schemas.BifrostContext) string {
+ return deriveAnthropicBaseURL(provider.networkConfig.BaseURL) + providerUtils.GetPathFromContext(ctx, anthropicMessagesPath)
+}
+
+// GetProviderKey returns the provider identifier for Alibaba Cloud Model Studio.
+func (provider *AlibabaProvider) GetProviderKey() schemas.ModelProvider {
+ return schemas.Alibaba
+}
+
+// ListModels performs a list models request to Model Studio's OpenAI-compatible API.
+// (The Anthropic mount has no /v1/models endpoint.)
+func (provider *AlibabaProvider) ListModels(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) {
+ return openai.HandleOpenAIListModelsRequest(
+ ctx,
+ provider.client,
+ request,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, modelsPath),
+ keys,
+ provider.networkConfig.ExtraHeaders,
+ provider.GetProviderKey(),
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ )
+}
+
+// TextCompletion is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionRequest, provider.GetProviderKey())
+}
+
+// TextCompletionStream is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionStreamRequest, provider.GetProviderKey())
+}
+
+// ChatCompletion performs a chat completion request to Model Studio's API.
+func (provider *AlibabaProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ return anthropic.HandleAnthropicChatCompletionRequest(
+ ctx,
+ provider.client,
+ provider.anthropicMessagesURL(ctx),
+ request,
+ anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Alibaba,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ },
+ provider.xAPIKeyHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ nil,
+ provider.logger,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIChatCompletionRequest(
+ ctx,
+ provider.client,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, chatCompletionsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ )
+}
+
+// ChatCompletionStream performs a streaming chat completion request to Model Studio's API.
+// It supports real-time streaming of responses using Server-Sent Events (SSE).
+// Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails.
+func (provider *AlibabaProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ jsonData, bifrostErr := anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Alibaba,
+ IsStreaming: true,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ })
+ if bifrostErr != nil {
+ return nil, bifrostErr
+ }
+
+ return anthropic.HandleAnthropicChatCompletionStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.anthropicMessagesURL(ctx),
+ jsonData,
+ provider.xAPIKeyHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ provider.networkConfig.BetaHeaderOverrides,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ schemas.Alibaba,
+ postHookRunner,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIChatCompletionStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, chatCompletionsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ schemas.Alibaba,
+ postHookRunner,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+}
+
+// Responses performs a Responses API request against Model Studio.
+// With use_anthropic_endpoints set, the request is served by the Anthropic
+// (Messages-only) mount; otherwise the native OpenAI-mount Responses API is used.
+func (provider *AlibabaProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ return anthropic.HandleAnthropicResponsesRequest(
+ ctx,
+ provider.client,
+ provider.anthropicMessagesURL(ctx),
+ request,
+ anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Alibaba,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ },
+ provider.xAPIKeyHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ nil,
+ provider.logger,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIResponsesRequest(
+ ctx,
+ provider.client,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, responsesPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ )
+}
+
+// ResponsesStream performs a streaming Responses API request to Model Studio.
+func (provider *AlibabaProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ jsonData, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Alibaba,
+ IsStreaming: true,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ })
+ if bifrostErr != nil {
+ return nil, bifrostErr
+ }
+
+ return anthropic.HandleAnthropicResponsesStream(
+ ctx,
+ provider.streamingClient,
+ provider.anthropicMessagesURL(ctx),
+ jsonData,
+ provider.xAPIKeyHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ provider.networkConfig.BetaHeaderOverrides,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ postHookRunner,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIResponsesStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, responsesPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ postHookRunner,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+}
+
+// Embedding performs an embedding request to Model Studio's OpenAI-compatible API
+// (text-embedding-v4 / text-embedding-v3).
+func (provider *AlibabaProvider) Embedding(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) {
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIEmbeddingRequest(
+ ctx,
+ provider.client,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, embeddingsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.GetProviderKey(),
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ nil,
+ provider.logger,
+ )
+}
+
+// Speech is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) Speech(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostSpeechRequest) (*schemas.BifrostSpeechResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechRequest, provider.GetProviderKey())
+}
+
+// Rerank is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostRerankRequest) (*schemas.BifrostRerankResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.RerankRequest, provider.GetProviderKey())
+}
+
+// OCR is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) OCR(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostOCRRequest) (*schemas.BifrostOCRResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.OCRRequest, provider.GetProviderKey())
+}
+
+// SpeechStream is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) SpeechStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostSpeechRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechStreamRequest, provider.GetProviderKey())
+}
+
+// Transcription is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) Transcription(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTranscriptionRequest) (*schemas.BifrostTranscriptionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionRequest, provider.GetProviderKey())
+}
+
+// TranscriptionStream is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) TranscriptionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTranscriptionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionStreamRequest, provider.GetProviderKey())
+}
+
+// ImageGeneration is not supported by the Alibaba provider (image models use
+// dedicated non-chat endpoints upstream).
+func (provider *AlibabaProvider) ImageGeneration(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageGenerationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationRequest, provider.GetProviderKey())
+}
+
+// ImageGenerationStream is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ImageGenerationStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageGenerationRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationStreamRequest, provider.GetProviderKey())
+}
+
+// ImageEdit is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ImageEdit(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageEditRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditRequest, provider.GetProviderKey())
+}
+
+// ImageEditStream is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ImageEditStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageEditRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditStreamRequest, provider.GetProviderKey())
+}
+
+// ImageVariation is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ImageVariation(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageVariationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageVariationRequest, provider.GetProviderKey())
+}
+
+// VideoGeneration is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoGeneration(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoGenerationRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoGenerationRequest, provider.GetProviderKey())
+}
+
+// VideoRetrieve is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoRetrieve(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRetrieveRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRetrieveRequest, provider.GetProviderKey())
+}
+
+// VideoDownload is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoDownload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDownloadRequest) (*schemas.BifrostVideoDownloadResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDownloadRequest, provider.GetProviderKey())
+}
+
+// VideoDelete is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoDelete(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDeleteRequest) (*schemas.BifrostVideoDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDeleteRequest, provider.GetProviderKey())
+}
+
+// VideoList is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoList(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoListRequest) (*schemas.BifrostVideoListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoListRequest, provider.GetProviderKey())
+}
+
+// VideoEdit is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoEdit(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoEditRequest) (*schemas.BifrostVideoEditResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoEditRequest, provider.GetProviderKey())
+}
+
+// VideoRemix is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) VideoRemix(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRemixRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRemixRequest, provider.GetProviderKey())
+}
+
+// FileUpload is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) FileUpload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostFileUploadRequest) (*schemas.BifrostFileUploadResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileUploadRequest, provider.GetProviderKey())
+}
+
+// FileList is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) FileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileListRequest) (*schemas.BifrostFileListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileListRequest, provider.GetProviderKey())
+}
+
+// FileRetrieve is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) FileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileRetrieveRequest) (*schemas.BifrostFileRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileRetrieveRequest, provider.GetProviderKey())
+}
+
+// FileDelete is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) FileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileDeleteRequest) (*schemas.BifrostFileDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileDeleteRequest, provider.GetProviderKey())
+}
+
+// FileContent is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) FileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileContentRequest) (*schemas.BifrostFileContentResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileContentRequest, provider.GetProviderKey())
+}
+
+// BatchCreate is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) BatchCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostBatchCreateRequest) (*schemas.BifrostBatchCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCreateRequest, provider.GetProviderKey())
+}
+
+// BatchList is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) BatchList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchListRequest) (*schemas.BifrostBatchListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchListRequest, provider.GetProviderKey())
+}
+
+// BatchRetrieve is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) BatchRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchRetrieveRequest) (*schemas.BifrostBatchRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchRetrieveRequest, provider.GetProviderKey())
+}
+
+// BatchCancel is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) BatchCancel(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchCancelRequest) (*schemas.BifrostBatchCancelResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCancelRequest, provider.GetProviderKey())
+}
+
+// BatchDelete is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) BatchDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchDeleteRequest) (*schemas.BifrostBatchDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchDeleteRequest, provider.GetProviderKey())
+}
+
+// BatchResults is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) BatchResults(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchResultsRequest) (*schemas.BifrostBatchResultsResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchResultsRequest, provider.GetProviderKey())
+}
+
+// CountTokens is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) CountTokens(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostCountTokensResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CountTokensRequest, provider.GetProviderKey())
+}
+
+// Compaction is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) Compaction(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCompactionRequest) (*schemas.BifrostCompactionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CompactionRequest, provider.GetProviderKey())
+}
+
+// ContainerCreate is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerCreateRequest) (*schemas.BifrostContainerCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerCreateRequest, provider.GetProviderKey())
+}
+
+// ContainerList is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerListRequest) (*schemas.BifrostContainerListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerListRequest, provider.GetProviderKey())
+}
+
+// ContainerRetrieve is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerRetrieveRequest) (*schemas.BifrostContainerRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerRetrieveRequest, provider.GetProviderKey())
+}
+
+// ContainerDelete is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerDeleteRequest) (*schemas.BifrostContainerDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerDeleteRequest, provider.GetProviderKey())
+}
+
+// ContainerFileCreate is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerFileCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerFileCreateRequest) (*schemas.BifrostContainerFileCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileCreateRequest, provider.GetProviderKey())
+}
+
+// ContainerFileList is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerFileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileListRequest) (*schemas.BifrostContainerFileListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileListRequest, provider.GetProviderKey())
+}
+
+// ContainerFileRetrieve is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerFileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileRetrieveRequest) (*schemas.BifrostContainerFileRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileRetrieveRequest, provider.GetProviderKey())
+}
+
+// ContainerFileContent is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerFileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileContentRequest) (*schemas.BifrostContainerFileContentResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileContentRequest, provider.GetProviderKey())
+}
+
+// ContainerFileDelete is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) ContainerFileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileDeleteRequest) (*schemas.BifrostContainerFileDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileDeleteRequest, provider.GetProviderKey())
+}
+
+// Passthrough is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) Passthrough(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (*schemas.BifrostPassthroughResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughRequest, provider.GetProviderKey())
+}
+
+// PassthroughStream is not supported by the Alibaba provider.
+func (provider *AlibabaProvider) PassthroughStream(_ *schemas.BifrostContext, _ schemas.PostHookRunner, _ func(context.Context), _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughStreamRequest, provider.GetProviderKey())
+}
diff --git a/core/providers/alibaba/alibaba_test.go b/core/providers/alibaba/alibaba_test.go
new file mode 100644
index 00000000000..5153d51916e
--- /dev/null
+++ b/core/providers/alibaba/alibaba_test.go
@@ -0,0 +1,63 @@
+package alibaba_test
+
+import (
+ "os"
+ "strings"
+ "testing"
+
+ "github.com/maximhq/bifrost/core/internal/llmtests"
+
+ "github.com/maximhq/bifrost/core/schemas"
+)
+
+func TestAlibaba(t *testing.T) {
+ t.Parallel()
+ if strings.TrimSpace(os.Getenv("ALIBABA_API_KEY")) == "" {
+ t.Skip("Skipping Alibaba tests because ALIBABA_API_KEY is not set")
+ }
+
+ client, ctx, cancel, err := llmtests.SetupTest()
+ if err != nil {
+ t.Fatalf("Error initializing test setup: %v", err)
+ }
+ defer cancel()
+ defer client.Shutdown()
+
+ testConfig := llmtests.ComprehensiveTestConfig{
+ Provider: schemas.Alibaba,
+ ChatModel: "qwen-flash", // cheap; thinking OFF by default (thinking-on models burn tight harness token caps on reasoning)
+ Fallbacks: []schemas.Fallback{
+ {Provider: schemas.Alibaba, Model: "qwen-flash"},
+ {Provider: schemas.Alibaba, Model: "qwen-turbo"},
+ },
+ TextModel: "qwen-flash",
+ EmbeddingModel: "text-embedding-v4",
+ ReasoningModel: "qwen3.7-plus", // Responses mount: accepts reasoning.effort high/xhigh (qwen3.6-flash 400s on them — vendor budget-mapping quirk)
+ Scenarios: llmtests.TestScenarios{
+ SimpleChat: true,
+ CompletionStream: true,
+ MultiTurnConversation: true,
+ ToolCalls: true,
+ ToolCallsStreaming: true,
+ AutomaticFunctionCall: true,
+ ImageURL: false,
+ ImageBase64: false,
+ MultipleImages: false,
+ // DashScope's /responses mount cannot ingest tool results: its agent
+ // backend rewrites function_call_output items to role:"tool" messages and
+ // then rejects them ("tool must be one of user,assistant,system,function"),
+ // while the outer layer rejects role:"function". Dual-API tool-continuation
+ // scenarios would fail on the Responses leg for every model — vendor gap.
+ CompleteEnd2End: false,
+ End2EndToolCalling: false,
+ Embedding: true,
+ ListModels: true,
+ Reasoning: true,
+ PassThroughExtraParams: true,
+ },
+ }
+
+ t.Run("AlibabaTests", func(t *testing.T) {
+ llmtests.RunAllComprehensiveTests(t, client, ctx, testConfig)
+ })
+}
diff --git a/core/providers/alibaba/cachedcontents.go b/core/providers/alibaba/cachedcontents.go
new file mode 100644
index 00000000000..4af31b6f9dd
--- /dev/null
+++ b/core/providers/alibaba/cachedcontents.go
@@ -0,0 +1,34 @@
+package alibaba
+
+import (
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ "github.com/maximhq/bifrost/core/schemas"
+)
+
+// CachedContentCreate is unsupported on AlibabaProvider. Only Gemini and Vertex AI
+// implement the cached-content lifecycle (Google AI Studio + Vertex AI named
+// caches). Model Studio handles caching via implicit caching plus content-level
+// cache_control markers.
+func (provider *AlibabaProvider) CachedContentCreate(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCachedContentCreateRequest) (*schemas.BifrostCachedContentCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentCreateRequest, provider.GetProviderKey())
+}
+
+// CachedContentList is unsupported on AlibabaProvider (see CachedContentCreate).
+func (provider *AlibabaProvider) CachedContentList(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentListRequest) (*schemas.BifrostCachedContentListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentListRequest, provider.GetProviderKey())
+}
+
+// CachedContentRetrieve is unsupported on AlibabaProvider (see CachedContentCreate).
+func (provider *AlibabaProvider) CachedContentRetrieve(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentRetrieveRequest) (*schemas.BifrostCachedContentRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentRetrieveRequest, provider.GetProviderKey())
+}
+
+// CachedContentUpdate is unsupported on AlibabaProvider (see CachedContentCreate).
+func (provider *AlibabaProvider) CachedContentUpdate(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentUpdateRequest) (*schemas.BifrostCachedContentUpdateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentUpdateRequest, provider.GetProviderKey())
+}
+
+// CachedContentDelete is unsupported on AlibabaProvider (see CachedContentCreate).
+func (provider *AlibabaProvider) CachedContentDelete(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentDeleteRequest) (*schemas.BifrostCachedContentDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentDeleteRequest, provider.GetProviderKey())
+}
diff --git a/core/providers/alibaba/utils.go b/core/providers/alibaba/utils.go
new file mode 100644
index 00000000000..c4e9e0354ac
--- /dev/null
+++ b/core/providers/alibaba/utils.go
@@ -0,0 +1,89 @@
+package alibaba
+
+import (
+ "net/url"
+ "strings"
+)
+
+const (
+ // defaultBaseURL is the Alibaba Cloud Model Studio international (Singapore) legacy
+ // host, OpenAI-compatible mount. Token Plan users override it with
+ // https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1, and
+ // workspace-dedicated hosts follow https://{WorkspaceId}.{region}.maas.aliyuncs.com/compatible-mode/v1.
+ defaultBaseURL = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
+
+ // chatCompletionsPath is the chat completions path relative to the OpenAI-compatible base URL.
+ chatCompletionsPath = "/chat/completions"
+
+ // modelsPath is the list-models path relative to the OpenAI-compatible base URL.
+ modelsPath = "/models"
+
+ // responsesPath is the Responses API path relative to the OpenAI-compatible base URL.
+ responsesPath = "/responses"
+
+ // embeddingsPath is the embeddings path relative to the OpenAI-compatible base URL.
+ embeddingsPath = "/embeddings"
+
+ // compatibleModeSuffix is the pay-as-you-go / Token Plan OpenAI mount suffix.
+ compatibleModeSuffix = "/compatible-mode/v1"
+
+ // anthropicMount is the Anthropic mount suffix on all Model Studio hosts.
+ anthropicMount = "/apps/anthropic"
+
+ // anthropicMessagesPath is the messages path under the derived Anthropic mount base.
+ anthropicMessagesPath = "/v1/messages"
+)
+
+// isKnownAlibabaHost reports whether the base URL sits on one of Alibaba
+// Cloud's own Model Studio hosts. Every host shape — dashscope[-intl|-us],
+// coding[-intl].dashscope, and the {WorkspaceId}.{region}.maas workspace and
+// token-plan hosts — lives under aliyuncs.com, so the check is a domain-boundary
+// suffix match. Unparseable or scheme-less inputs are treated as custom hosts.
+func isKnownAlibabaHost(base string) bool {
+ parsed, err := url.Parse(base)
+ if err != nil || parsed.Scheme == "" || parsed.Host == "" {
+ return false
+ }
+ return parsed.Host == "aliyuncs.com" || strings.HasSuffix(parsed.Host, ".aliyuncs.com")
+}
+
+// deriveAnthropicBaseURL derives the Anthropic-compatible mount base URL from the
+// configured OpenAI-compatible base URL.
+//
+// Every Model Studio host shape mounts Messages at /apps/anthropic:
+//
+// - https://dashscope-intl.aliyuncs.com/compatible-mode/v1
+// -> https://dashscope-intl.aliyuncs.com/apps/anthropic
+// - https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1
+// -> https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic
+// - https://{WorkspaceId}.{region}.maas.aliyuncs.com/compatible-mode/v1
+// -> https://{WorkspaceId}.{region}.maas.aliyuncs.com/apps/anthropic
+// - Coding Plan hosts (coding[-intl].dashscope.aliyuncs.com/v1)
+// -> same host + /apps/anthropic
+//
+// The path rewrites only apply on Alibaba's own hosts (*.aliyuncs.com). Any other
+// base URL — including a custom or proxied base that merely ends in
+// /compatible-mode/v1 or /v1 — keeps its configured path and only gets
+// /apps/anthropic appended, so a foreign URL shape is never silently rewritten;
+// users on exotic hosts can always create a second provider instance with an
+// explicit base URL.
+//
+// Idempotent: a base that already ends with the mount suffix is returned unchanged —
+// use_anthropic_endpoints with a base_url set to the mount itself must not append
+// /apps/anthropic a second time.
+func deriveAnthropicBaseURL(openAIBaseURL string) string {
+ base := strings.TrimRight(openAIBaseURL, "/")
+ if strings.HasSuffix(base, anthropicMount) {
+ return base
+ }
+ if !isKnownAlibabaHost(base) {
+ return base + anthropicMount
+ }
+ if strings.HasSuffix(base, compatibleModeSuffix) {
+ return strings.TrimSuffix(base, compatibleModeSuffix) + anthropicMount
+ }
+ if strings.HasSuffix(base, "/v1") {
+ return strings.TrimSuffix(base, "/v1") + anthropicMount
+ }
+ return base + anthropicMount
+}
diff --git a/core/providers/alibaba/utils_test.go b/core/providers/alibaba/utils_test.go
new file mode 100644
index 00000000000..c4ea85198a8
--- /dev/null
+++ b/core/providers/alibaba/utils_test.go
@@ -0,0 +1,105 @@
+package alibaba
+
+import "testing"
+
+func TestDeriveAnthropicBaseURL(t *testing.T) {
+ tests := []struct {
+ name string
+ openAIBase string
+ wantMessages string // full messages URL
+ }{
+ {
+ name: "legacy shared international (Singapore)",
+ openAIBase: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
+ wantMessages: "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "legacy shared cn-beijing",
+ openAIBase: "https://dashscope.aliyuncs.com/compatible-mode/v1",
+ wantMessages: "https://dashscope.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "legacy shared us-east-1",
+ openAIBase: "https://dashscope-us.aliyuncs.com/compatible-mode/v1",
+ wantMessages: "https://dashscope-us.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "workspace-dedicated host",
+ openAIBase: "https://ws-abc123.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
+ wantMessages: "https://ws-abc123.ap-southeast-1.maas.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "Token Plan Team international",
+ openAIBase: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
+ wantMessages: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "Coding Plan China host",
+ openAIBase: "https://coding.dashscope.aliyuncs.com/v1",
+ wantMessages: "https://coding.dashscope.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "Coding Plan international host",
+ openAIBase: "https://coding-intl.dashscope.aliyuncs.com/v1",
+ wantMessages: "https://coding-intl.dashscope.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "trailing slash is trimmed",
+ openAIBase: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/",
+ wantMessages: "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "custom base falls back to appending /apps/anthropic",
+ openAIBase: "https://proxy.example.com/qwen",
+ wantMessages: "https://proxy.example.com/qwen/apps/anthropic/v1/messages",
+ },
+ {
+ // Suffix semantics only hold on Alibaba's own hosts: a custom host
+ // ending in /compatible-mode/v1 must NOT have it rewritten away.
+ name: "custom base ending in /compatible-mode/v1 keeps its path",
+ openAIBase: "https://proxy.example.com/compatible-mode/v1",
+ wantMessages: "https://proxy.example.com/compatible-mode/v1/apps/anthropic/v1/messages",
+ },
+ {
+ name: "custom base ending in /v1 keeps its path",
+ openAIBase: "https://proxy.example.com/v1",
+ wantMessages: "https://proxy.example.com/v1/apps/anthropic/v1/messages",
+ },
+ {
+ // The aliyuncs.com gate must respect the domain boundary — a host
+ // whose name merely CONTAINS aliyuncs.com is a foreign host.
+ name: "lookalike host ending in aliyuncs.com. keeps its path",
+ openAIBase: "https://dashscope-intl.aliyuncs.com.evil.example/compatible-mode/v1",
+ wantMessages: "https://dashscope-intl.aliyuncs.com.evil.example/compatible-mode/v1/apps/anthropic/v1/messages",
+ },
+ {
+ // Scheme-less values must take the custom-host fallback even when
+ // the remainder parses to a known Alibaba host.
+ name: "scheme-less base takes the custom-host fallback",
+ openAIBase: "//dashscope-intl.aliyuncs.com/compatible-mode/v1",
+ wantMessages: "//dashscope-intl.aliyuncs.com/compatible-mode/v1/apps/anthropic/v1/messages",
+ },
+ {
+ // use_anthropic_endpoints with a base_url already set to the mount
+ // itself must not append /apps/anthropic a second time.
+ name: "Token Plan host with the Anthropic mount as base is idempotent",
+ openAIBase: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic",
+ wantMessages: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ {
+ name: "Anthropic mount as base with trailing slash is idempotent",
+ openAIBase: "https://dashscope-intl.aliyuncs.com/apps/anthropic/",
+ wantMessages: "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
+ },
+ }
+
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ base := deriveAnthropicBaseURL(tt.openAIBase)
+ full := base + anthropicMessagesPath
+ if full != tt.wantMessages {
+ t.Errorf("messages URL for %q = %q, want %q", tt.openAIBase, full, tt.wantMessages)
+ }
+ })
+ }
+}
diff --git a/core/providers/anthropic/anthropic.go b/core/providers/anthropic/anthropic.go
index 231cb090dff..67337939d78 100644
--- a/core/providers/anthropic/anthropic.go
+++ b/core/providers/anthropic/anthropic.go
@@ -127,6 +127,53 @@ func (provider *AnthropicProvider) GetProviderKey() schemas.ModelProvider {
return providerUtils.GetProviderName(schemas.Anthropic, provider.customProviderConfig)
}
+// conversionProvider returns the capability profile used for request conversion
+// and feature gating. The stock provider keeps schemas.Anthropic; custom
+// providers (base_provider_type: anthropic) resolve theirs from the configured
+// base URL host (see ResolveAnthropicMountProfile), so a "zai-anthropic" mount
+// pointed at api.z.ai gets Zhipu's profile — output_config.effort passthrough
+// and GLM forced-thinking handling — instead of Anthropic's own model gates.
+func (provider *AnthropicProvider) conversionProvider() schemas.ModelProvider {
+ if provider.customProviderConfig == nil {
+ return schemas.Anthropic
+ }
+ return ResolveAnthropicMountProfile(provider.networkConfig.BaseURL)
+}
+
+// normalizeChatRequestForConversion returns the request unchanged unless the
+// resolved mount profile differs from schemas.Anthropic — then it returns a
+// shallow copy with Provider set to the profile so converter-level gates (which
+// key on bifrostReq.Provider) see the same profile the builder strips with.
+// The caller's request is never mutated. Mirrors Mistral's
+// normalizeChatRequestForConversion.
+func (provider *AnthropicProvider) normalizeChatRequestForConversion(request *schemas.BifrostChatRequest) *schemas.BifrostChatRequest {
+ if request == nil {
+ return request
+ }
+ profile := provider.conversionProvider()
+ if profile == schemas.Anthropic || request.Provider == profile {
+ return request
+ }
+ normalized := *request
+ normalized.Provider = profile
+ return &normalized
+}
+
+// normalizeResponsesRequestForConversion is the Responses-API analogue of
+// normalizeChatRequestForConversion.
+func (provider *AnthropicProvider) normalizeResponsesRequestForConversion(request *schemas.BifrostResponsesRequest) *schemas.BifrostResponsesRequest {
+ if request == nil {
+ return request
+ }
+ profile := provider.conversionProvider()
+ if profile == schemas.Anthropic || request.Provider == profile {
+ return request
+ }
+ normalized := *request
+ normalized.Provider = profile
+ return &normalized
+}
+
// buildRequestURL constructs the full request URL using the provider's configuration.
func (provider *AnthropicProvider) buildRequestURL(ctx *schemas.BifrostContext, defaultPath string, requestType schemas.RequestType) string {
path, isCompleteURL := providerUtils.GetRequestPath(ctx, defaultPath, provider.customProviderConfig, requestType)
@@ -482,9 +529,9 @@ func (provider *AnthropicProvider) ChatCompletion(ctx *schemas.BifrostContext, k
ctx,
provider.client,
provider.buildRequestURL(ctx, "/v1/messages", schemas.ChatCompletionRequest),
- request,
+ provider.normalizeChatRequestForConversion(request),
AnthropicRequestBuildConfig{
- Provider: schemas.Anthropic,
+ Provider: provider.conversionProvider(),
IsStreaming: false,
BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides,
ShouldSendBackRawRequest: provider.sendBackRawRequest,
@@ -577,8 +624,8 @@ func (provider *AnthropicProvider) ChatCompletionStream(ctx *schemas.BifrostCont
return nil, err
}
- jsonData, bifrostErr := BuildAnthropicChatRequestBody(ctx, request, AnthropicRequestBuildConfig{
- Provider: schemas.Anthropic,
+ jsonData, bifrostErr := BuildAnthropicChatRequestBody(ctx, provider.normalizeChatRequestForConversion(request), AnthropicRequestBuildConfig{
+ Provider: provider.conversionProvider(),
IsStreaming: true,
ShouldSendBackRawRequest: provider.sendBackRawRequest,
ShouldSendBackRawResponse: provider.sendBackRawResponse,
@@ -1277,9 +1324,9 @@ func (provider *AnthropicProvider) Responses(ctx *schemas.BifrostContext, key sc
ctx,
provider.client,
provider.buildRequestURL(ctx, "/v1/messages", schemas.ResponsesRequest),
- request,
+ provider.normalizeResponsesRequestForConversion(request),
AnthropicRequestBuildConfig{
- Provider: schemas.Anthropic,
+ Provider: provider.conversionProvider(),
IsStreaming: false,
BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides,
ShouldSendBackRawRequest: provider.sendBackRawRequest,
@@ -1368,8 +1415,8 @@ func (provider *AnthropicProvider) ResponsesStream(ctx *schemas.BifrostContext,
return nil, err
}
- jsonBody, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{
- Provider: schemas.Anthropic,
+ jsonBody, err := BuildAnthropicResponsesRequestBody(ctx, provider.normalizeResponsesRequestForConversion(request), AnthropicRequestBuildConfig{
+ Provider: provider.conversionProvider(),
IsStreaming: true,
ShouldSendBackRawRequest: provider.sendBackRawRequest,
ShouldSendBackRawResponse: provider.sendBackRawResponse,
diff --git a/core/providers/anthropic/chat.go b/core/providers/anthropic/chat.go
index fec4ce1eb09..2bbb6193e4d 100644
--- a/core/providers/anthropic/chat.go
+++ b/core/providers/anthropic/chat.go
@@ -694,23 +694,45 @@ func ToAnthropicChatRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.Bif
Type: "enabled",
BudgetTokens: schemas.Ptr(budgetTokens),
}
+ // Vendor extension mounts keep a co-present effort instead
+ // of discarding it. z.ai accepts thinking.budget_tokens and
+ // output_config.effort together (the ZCode-proven shape);
+ // Model Studio rejects the pair ("'reasoning_effort' and
+ // 'thinking_budget' cannot be set simultaneously") and
+ // engages thinking itself from the effort value, so the
+ // effort wins there and the thinking field is dropped
+ // (verified live 2026-08-23).
+ if reasoningParams.Effort != nil && *reasoningParams.Effort != "none" &&
+ SupportsProviderEffort(bifrostReq.Provider, capModel) {
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, MapBifrostEffortToAnthropic(*reasoningParams.Effort))
+ if bifrostReq.Provider == schemas.Alibaba {
+ anthropicReq.Thinking = nil
+ }
+ }
}
} else if reasoningParams.Effort != nil && *reasoningParams.Effort != "none" {
effort := MapBifrostEffortToAnthropic(*reasoningParams.Effort)
if caps.SupportsAdaptiveThinking(DefaultSupportsAdaptiveThinking(caps.Model())) {
// Opus 4.6+ and Opus 4.7+: adaptive thinking + native effort
anthropicReq.Thinking = &AnthropicThinking{Type: "adaptive"}
- setEffortOnOutputConfig(anthropicReq, effort)
- } else if SupportsNativeEffort(caps) {
- // Opus 4.5: native effort + budget_tokens thinking
- setEffortOnOutputConfig(anthropicReq, effort)
- budgetTokens, err := providerUtils.GetBudgetTokensFromReasoningEffort(effort, MinimumReasoningMaxTokens, anthropicReq.MaxTokens)
- if err != nil {
- return nil, fmt.Errorf("%w: %w", ErrReasoningMaxTokensTooLow, err)
- }
- anthropicReq.Thinking = &AnthropicThinking{
- Type: "enabled",
- BudgetTokens: schemas.Ptr(budgetTokens),
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, effort)
+ } else if SupportsNativeEffort(caps) || SupportsProviderEffort(bifrostReq.Provider, capModel) {
+ // Opus 4.5: native effort + budget_tokens thinking.
+ // z.ai (GLM-5.2+) takes the same shape — the mount maps the
+ // effort value server-side. Model Studio (Alibaba) rejects
+ // effort + thinking_budget together and engages thinking
+ // itself from the effort value, so the effort is forwarded
+ // alone there (verified live 2026-08-23).
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, effort)
+ if bifrostReq.Provider != schemas.Alibaba {
+ budgetTokens, err := providerUtils.GetBudgetTokensFromReasoningEffort(effort, MinimumReasoningMaxTokens, anthropicReq.MaxTokens)
+ if err != nil {
+ return nil, fmt.Errorf("%w: %w", ErrReasoningMaxTokensTooLow, err)
+ }
+ anthropicReq.Thinking = &AnthropicThinking{
+ Type: "enabled",
+ BudgetTokens: schemas.Ptr(budgetTokens),
+ }
}
} else {
// Older models: budget_tokens only
@@ -727,7 +749,9 @@ func ToAnthropicChatRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.Bif
// Fable/Mythos reject thinking:{type:"disabled"} with a 400 —
// adaptive thinking is always on and cannot be disabled. Omit
// the thinking param entirely for that family; all other models
- // take the explicit disabled path.
+ // take the explicit disabled path. (Forced-thinking GLM models
+ // get rewritten to enabled downstream in
+ // stripUnsupportedAnthropicFields.)
anthropicReq.Thinking = &AnthropicThinking{
Type: "disabled",
}
diff --git a/core/providers/anthropic/providereffort_test.go b/core/providers/anthropic/providereffort_test.go
new file mode 100644
index 00000000000..0f12b099824
--- /dev/null
+++ b/core/providers/anthropic/providereffort_test.go
@@ -0,0 +1,951 @@
+package anthropic
+
+import (
+ "context"
+ "testing"
+
+ "github.com/maximhq/bifrost/core/providers/openai"
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ "github.com/maximhq/bifrost/core/schemas"
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+)
+
+// Regression coverage for output_config.effort on non-Anthropic
+// Anthropic-compatible mounts, verified against vendor docs 2026-08-16:
+//
+// - z.ai Coding Plan Anthropic mount (schemas.Zhipu): output_config.effort is a
+// documented extension for GLM-5.2+; out-of-scale values are mapped
+// server-side (GLM-5.3: none/minimal/low→low, medium/high→high,
+// xhigh/max→max). thinking.type:"disabled" 400s on GLM-5.3+ and is ignored
+// on forced-thinking GLM-4.7, so the field is omitted instead.
+// Source: https://docs.z.ai/guides/capabilities/thinking
+// - Alibaba Cloud Model Studio /apps/anthropic (schemas.Alibaba):
+// output_config.effort documented for qwen3.8-max, hosted glm-5.2 and
+// deepseek-v4-pro/flash. The mount does NOT map out-of-enum values
+// server-side: it validates the field against the per-model
+// reasoning_effort enum and 400s on anything else ("'reasoning_effort'
+// must be one of: ...", verified live 2026-08-23), so the gateway emits
+// only each family's officially valid values (see
+// clampAlibabaMountEffortForModel for the per-model matrix; vendor API
+// docs supplied 2026-08-23). The mount also rejects effort and
+// thinking_budget set together and engages thinking on its own from the
+// effort value, so a forwarded effort travels alone (no synthesized
+// thinking field).
+// Source: https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages
+// - Kimi /anthropic (schemas.Kimi): unofficial empirical contract
+// (MoonshotAI/Kimi-K2#129) — no effort equivalent documented, so effort
+// keeps being stripped rather than forwarded as an unknown field.
+//
+// The reported bug: an Anthropic-protocol client sending
+// {"output_config":{"effort":"max"}} with no thinking parameter had the effort
+// silently discarded on the zhipu mount because every gate was keyed on
+// Anthropic's own model list.
+
+func providerEffortTestCtx(t *testing.T) *schemas.BifrostContext {
+ t.Helper()
+ ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background())
+ t.Cleanup(cancel)
+ return ctx
+}
+
+// anthropicInbound builds the Messages API shape an Anthropic-protocol client
+// (opencode / Claude Code style) sends, then normalizes it the way the inbound
+// integration does.
+func anthropicInbound(model string, effort *string, thinking *AnthropicThinking) *AnthropicMessageRequest {
+ req := &AnthropicMessageRequest{
+ Model: model,
+ MaxTokens: 128000,
+ Messages: []AnthropicMessage{{
+ Role: AnthropicMessageRoleUser,
+ Content: AnthropicContent{ContentStr: schemas.Ptr("What is 1+1?")},
+ }},
+ Thinking: thinking,
+ }
+ if effort != nil {
+ req.OutputConfig = &AnthropicOutputConfig{Effort: effort}
+ }
+ return req
+}
+
+func TestSupportsProviderEffort(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ provider schemas.ModelProvider
+ model string
+ want bool
+ }{
+ // Zhipu: GLM-5.2+ only (z.ai documents reasoning_effort for "GLM-5.2 and above").
+ {schemas.Zhipu, "glm-5.3", true},
+ {schemas.Zhipu, "GLM-5.3", true},
+ {schemas.Zhipu, "glm-5.2", true},
+ {schemas.Zhipu, "glm-5.2[1m]", true},
+ {schemas.Zhipu, "glm-5", false},
+ {schemas.Zhipu, "glm-5.1", false},
+ {schemas.Zhipu, "glm-4.7", false},
+ {schemas.Zhipu, "glm-4.6v", false},
+ // Alibaba: qwen3.8-max / hosted glm-5.2+ / deepseek-v4 only.
+ {schemas.Alibaba, "qwen3.8-max", true},
+ {schemas.Alibaba, "qwen3.8-max-preview", true},
+ {schemas.Alibaba, "glm-5.2", true},
+ {schemas.Alibaba, "deepseek-v4-pro", true},
+ {schemas.Alibaba, "deepseek-v4-flash", true},
+ {schemas.Alibaba, "deepseek-v4-flash-0731", true},
+ {schemas.Alibaba, "qwen3.7-plus", false},
+ {schemas.Alibaba, "qwen3-max", false},
+ {schemas.Alibaba, "glm-5.1", false},
+ {schemas.Alibaba, "MiniMax-M2.5", false},
+ // Kimi: no effort equivalent on the /anthropic mount.
+ {schemas.Kimi, "kimi-k3", false},
+ {schemas.Kimi, "kimi-k2.6", false},
+ // Anthropic falls back to the model gate unchanged.
+ {schemas.Anthropic, "claude-opus-4-6", true},
+ {schemas.Anthropic, "claude-sonnet-4-5", false},
+ {schemas.Anthropic, "claude-haiku-4-5", false},
+ // Unknown provider: model gate only (safe default, unchanged behavior).
+ {schemas.ModelProvider("unknown-vendor"), "claude-opus-4-6", true},
+ {schemas.ModelProvider("unknown-vendor"), "glm-5.3", false},
+ }
+
+ for _, tc := range cases {
+ t.Run(string(tc.provider)+"/"+tc.model, func(t *testing.T) {
+ assert.Equal(t, tc.want, SupportsProviderEffort(tc.provider, tc.model),
+ "SupportsProviderEffort(%s, %s)", tc.provider, tc.model)
+ })
+ }
+}
+
+func TestZhipuForcedThinkingModel(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ model string
+ want bool
+ }{
+ {"glm-5.3", true},
+ {"GLM-5.3", true},
+ {"glm-5.3[1m]", true},
+ {"glm-5.4", true},
+ {"glm-4.7", true},
+ {"glm-4.7-flash", true},
+ {"glm-4.5v", true},
+ {"glm-5.2", false},
+ {"glm-5.2[1m]", false},
+ {"glm-5", false},
+ {"glm-4.6", false},
+ {"qwen3.8-max", false},
+ }
+ for _, tc := range cases {
+ t.Run(tc.model, func(t *testing.T) {
+ assert.Equal(t, tc.want, ZhipuForcedThinkingModel(tc.model))
+ })
+ }
+}
+
+func TestZhipuRequiresThinkingModel(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ model string
+ want bool
+ }{
+ {"glm-5.3", true},
+ {"glm-5.3[1m]", true},
+ {"glm-5.4", true},
+ // GLM-4.7 tolerates an absent thinking field (Claude Code's default
+ // traffic against the mount carries none), so it is NOT here.
+ {"glm-4.7", false},
+ {"glm-4.5v", false},
+ {"glm-5.2", false},
+ {"glm-5", false},
+ }
+ for _, tc := range cases {
+ t.Run(tc.model, func(t *testing.T) {
+ assert.Equal(t, tc.want, ZhipuRequiresThinkingModel(tc.model))
+ })
+ }
+}
+
+// toAnthropicResponsesBuilt mirrors the builder flow: convert, then run the
+// strip pass the builder applies (ToAnthropicResponsesRequest does not strip
+// internally; BuildAnthropicResponsesRequestBody does).
+func toAnthropicResponsesBuilt(t *testing.T, ctx *schemas.BifrostContext, bifrostReq *schemas.BifrostResponsesRequest) *AnthropicMessageRequest {
+ t.Helper()
+ out, err := ToAnthropicResponsesRequest(ctx, bifrostReq)
+ require.NoError(t, err)
+ require.NotNil(t, out)
+ stripUnsupportedAnthropicFields(out, bifrostReq.Provider, out.Model)
+ return out
+}
+
+// TestZhipuAnthropicMount_EffortOnlyRoundTrip is the direct regression for the
+// reported bug: an Anthropic-protocol client sending output_config.effort with
+// no thinking parameter must see the effort reach the z.ai upstream verbatim.
+// GLM-5.3 also requires the thinking field outright on this mount (absent =
+// disabled = 1210 error, verified live 2026-08-16), so a thinking:{enabled} is
+// synthesized with the budget derived from the caller's effort.
+func TestZhipuAnthropicMount_EffortOnlyRoundTrip(t *testing.T) {
+ t.Parallel()
+
+ for _, effort := range []string{"max", "high", "low"} {
+ t.Run(effort, func(t *testing.T) {
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("glm-5.3", schemas.Ptr(effort), nil).ToBifrostResponsesRequest(ctx)
+ require.NotNil(t, bifrostReq)
+ bifrostReq.Provider = schemas.Zhipu
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ require.NotNil(t, out.OutputConfig, "output_config was dropped on the zhipu mount")
+ require.NotNil(t, out.OutputConfig.Effort)
+ assert.Equal(t, effort, *out.OutputConfig.Effort, "effort must reach z.ai verbatim (Coding Plan maps server-side)")
+ require.NotNil(t, out.Thinking, "GLM-5.3 rejects requests with no thinking field on this mount")
+ assert.Equal(t, "enabled", out.Thinking.Type)
+ require.NotNil(t, out.Thinking.BudgetTokens)
+ assert.Positive(t, *out.Thinking.BudgetTokens)
+ })
+ }
+}
+
+// TestZhipuAnthropicMount_CoPresentBudgetAndEffortRoundTrip pins the ZCode-proven
+// wire shape: thinking{budget_tokens} and output_config{effort} coexist.
+func TestZhipuAnthropicMount_CoPresentBudgetAndEffortRoundTrip(t *testing.T) {
+ t.Parallel()
+
+ ctx := providerEffortTestCtx(t)
+ budget := 32000
+ bifrostReq := anthropicInbound("glm-5.3", schemas.Ptr("max"), &AnthropicThinking{
+ Type: "enabled",
+ BudgetTokens: &budget,
+ }).ToBifrostResponsesRequest(ctx)
+ require.NotNil(t, bifrostReq)
+ bifrostReq.Provider = schemas.Zhipu
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ require.NotNil(t, out.Thinking)
+ assert.Equal(t, "enabled", out.Thinking.Type)
+ require.NotNil(t, out.Thinking.BudgetTokens)
+ assert.Equal(t, 32000, *out.Thinking.BudgetTokens)
+ require.NotNil(t, out.OutputConfig)
+ require.NotNil(t, out.OutputConfig.Effort)
+ assert.Equal(t, "max", *out.OutputConfig.Effort)
+}
+
+// TestZhipuAnthropicMount_NoReasoningSignalSynthesizesThinking covers a plain
+// request (no effort, no thinking) against GLM-5.3: the mount treats an absent
+// thinking field as disabled and 400s, so the strip pass synthesizes
+// thinking:{enabled} with the minimum budget.
+func TestZhipuAnthropicMount_NoReasoningSignalSynthesizesThinking(t *testing.T) {
+ t.Parallel()
+
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("glm-5.3", nil, nil).ToBifrostResponsesRequest(ctx)
+ require.NotNil(t, bifrostReq)
+ bifrostReq.Provider = schemas.Zhipu
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ require.NotNil(t, out.Thinking, "GLM-5.3 requires an explicit thinking field on z.ai's mount")
+ assert.Equal(t, "enabled", out.Thinking.Type)
+ require.NotNil(t, out.Thinking.BudgetTokens)
+ assert.Equal(t, MinimumReasoningMaxTokens, *out.Thinking.BudgetTokens)
+}
+
+// TestZhipuAnthropicMount_EffortDroppedBelowGLM52 pins the documented model
+// scope: GLM-5/5.1/4.7 do not accept effort, so it is stripped (budget thinking
+// still carries the intent) rather than risking an upstream 400.
+func TestZhipuAnthropicMount_EffortDroppedBelowGLM52(t *testing.T) {
+ t.Parallel()
+
+ for _, model := range []string{"glm-5", "glm-5.1", "glm-4.7"} {
+ t.Run(model, func(t *testing.T) {
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound(model, schemas.Ptr("max"), nil).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Zhipu
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ if out.OutputConfig != nil {
+ assert.Nil(t, out.OutputConfig.Effort, "%s does not accept output_config.effort", model)
+ }
+ // GLM-4.7 tolerates an absent thinking field, and glm-5/5.1 allow
+ // disabled — none of them require a synthesized thinking field.
+ assert.Nil(t, out.Thinking)
+ })
+ }
+}
+
+// TestAlibabaAnthropicMount_EffortRoundTrip covers the Model Studio families
+// with documented effort support. The mount validates the value against its
+// per-model reasoning_effort enum instead of mapping it server-side (verified
+// live 2026-08-23: a forwarded "max" drew "'reasoning_effort' must be one of:
+// ..."), so the gateway emits only each family's officially valid values
+// (vendor API docs supplied 2026-08-23; see clampAlibabaMountEffortForModel):
+//
+// - qwen3.8-max: valid xhigh/medium/low; max clamps to xhigh.
+// - glm-5.2/5.1/5 and non-dated deepseek-v4-pro/flash: valid high/max;
+// xhigh→max, low/medium→high, minimal/none→high.
+// - glm-5.3+: valid max/high/low; xhigh→max, medium→high, minimal/none→low.
+// - deepseek-v4-pro-0813 / deepseek-v4-flash-0731: valid max/high/low;
+// xhigh/medium→high, minimal/none→low.
+func TestAlibabaAnthropicMount_EffortRoundTrip(t *testing.T) {
+ t.Parallel()
+
+ supported := []struct{ model, effort, want string }{
+ {"qwen3.8-max", "xhigh", "xhigh"},
+ {"qwen3.8-max", "high", "high"},
+ {"qwen3.8-max", "medium", "medium"},
+ {"qwen3.8-max", "low", "low"},
+ {"qwen3.8-max", "max", "xhigh"}, // out of the mount enum; gateway clamps to the top tier
+ {"glm-5.2", "max", "max"},
+ {"glm-5.2", "xhigh", "max"},
+ {"glm-5.2", "high", "high"},
+ {"glm-5.2", "low", "high"},
+ {"glm-5.2", "medium", "high"},
+ // "minimal" is covered in TestClampAlibabaMountEffortForModel: the
+ // conversion pre-maps minimal→low (MapBifrostEffortToAnthropic) before
+ // the clamp runs, so on this wire path it lands on "high" — the
+ // clamp's own minimal→high rule applies to raw/native bodies.
+ {"glm-5.3", "max", "max"},
+ {"glm-5.3", "high", "high"},
+ {"glm-5.3", "low", "low"},
+ {"glm-5.3", "xhigh", "max"},
+ {"glm-5.3", "medium", "high"},
+ {"glm-5.3", "none", "low"},
+ {"deepseek-v4-pro", "max", "max"},
+ {"deepseek-v4-pro", "xhigh", "max"},
+ {"deepseek-v4-pro-0813", "max", "max"},
+ {"deepseek-v4-pro-0813", "xhigh", "high"},
+ {"deepseek-v4-flash-0731", "medium", "high"},
+ // Case variants: provider-prefixed / mixed-case model strings.
+ {"ZHIPU/GLM-5.3", "xhigh", "max"},
+ {"alibaba/glm-5.2", "max", "max"},
+ }
+ for _, tc := range supported {
+ t.Run(tc.model+"/"+tc.effort, func(t *testing.T) {
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound(tc.model, schemas.Ptr(tc.effort), nil).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Alibaba
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ require.NotNil(t, out.OutputConfig, "output_config was dropped on the alibaba mount")
+ require.NotNil(t, out.OutputConfig.Effort)
+ assert.Equal(t, tc.want, *out.OutputConfig.Effort)
+ // Alibaba treats an absent thinking field as the model default and
+ // rejects effort + thinking_budget together, so nothing is
+ // synthesized alongside a forwarded effort.
+ assert.Nil(t, out.Thinking)
+ })
+ }
+
+ // Families without documented effort support keep the strip behavior.
+ for _, model := range []string{"qwen3.7-plus", "qwen3-max", "glm-5.1"} {
+ t.Run(model+"-stripped", func(t *testing.T) {
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound(model, schemas.Ptr("max"), nil).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Alibaba
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ if out.OutputConfig != nil {
+ assert.Nil(t, out.OutputConfig.Effort, "%s has no documented effort support on Model Studio", model)
+ }
+ })
+ }
+}
+
+// TestClampAlibabaMountEffortForModel pins the full per-model clamp matrix on
+// the alibaba Anthropic mount directly (vendor API docs supplied 2026-08-23).
+// Every output is an officially valid value for the model family — the mount
+// validates against its per-model enum instead of mapping server-side.
+func TestClampAlibabaMountEffortForModel(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ model string
+ effort string
+ want string
+ }{
+ // qwen3.8-max: max→xhigh, everything else verbatim (unchanged).
+ {"qwen3.8-max", "max", "xhigh"},
+ {"qwen3.8-max", "xhigh", "xhigh"},
+ {"qwen3.8-max", "high", "high"},
+ {"qwen3.8-max", "low", "low"},
+ // glm-5.2/5.1: valid high/max; xhigh→max, low/medium→high, minimal/none→high.
+ {"glm-5.2", "max", "max"},
+ {"glm-5.2", "xhigh", "max"},
+ {"glm-5.2", "high", "high"},
+ {"glm-5.2", "low", "high"},
+ {"glm-5.2", "medium", "high"},
+ {"glm-5.2", "minimal", "high"},
+ {"glm-5.2", "none", "high"},
+ {"glm-5.1", "max", "max"},
+ {"glm-5.1", "xhigh", "max"},
+ // glm-5.3+: valid max/high/low; xhigh→max, medium→high, minimal/none→low.
+ {"glm-5.3", "max", "max"},
+ {"glm-5.3", "high", "high"},
+ {"glm-5.3", "low", "low"},
+ {"glm-5.3", "xhigh", "max"},
+ {"glm-5.3", "medium", "high"},
+ {"glm-5.3", "minimal", "low"},
+ {"glm-5.3", "none", "low"},
+ // Non-dated deepseek-v4: same family as glm-5.2 (valid high/max).
+ {"deepseek-v4-pro", "max", "max"},
+ {"deepseek-v4-pro", "xhigh", "max"},
+ {"deepseek-v4-pro", "low", "high"},
+ {"deepseek-v4-pro", "minimal", "high"},
+ {"deepseek-v4-flash", "medium", "high"},
+ // Dated snapshots: valid max/high/low; xhigh/medium→high, minimal/none→low.
+ {"deepseek-v4-pro-0813", "max", "max"},
+ {"deepseek-v4-pro-0813", "high", "high"},
+ {"deepseek-v4-pro-0813", "low", "low"},
+ {"deepseek-v4-pro-0813", "xhigh", "high"},
+ {"deepseek-v4-pro-0813", "medium", "high"},
+ {"deepseek-v4-pro-0813", "minimal", "low"},
+ {"deepseek-v4-flash-0731", "max", "max"},
+ {"deepseek-v4-flash-0731", "xhigh", "high"},
+ {"deepseek-v4-flash-0731", "medium", "high"},
+ {"deepseek-v4-flash-0731", "none", "low"},
+ // Case variants: provider-prefixed / mixed-case model strings.
+ {"ZHIPU/GLM-5.3", "xhigh", "max"},
+ {"ZHIPU/GLM-5.3", "medium", "high"},
+ {"alibaba/glm-5.2", "max", "max"},
+ {"alibaba/glm-5.2", "low", "high"},
+ // Anything else verbatim — kimi-k3 never reaches the clamp (its effort
+ // is stripped upstream by the provider effort gate).
+ {"kimi-k3", "max", "max"},
+ {"qwen3.7-plus", "max", "max"},
+ {"MiniMax-M2.5", "xhigh", "xhigh"},
+ }
+ for _, tc := range cases {
+ t.Run(tc.model+"/"+tc.effort, func(t *testing.T) {
+ assert.Equal(t, tc.want, clampAlibabaMountEffortForModel(tc.model, tc.effort),
+ "clampAlibabaMountEffortForModel(%s, %s)", tc.model, tc.effort)
+ })
+ }
+
+ // The provider gate: only the alibaba profile clamps; zhipu forwards
+ // verbatim (Coding Plan maps out-of-scale values server-side).
+ assert.Equal(t, "max", clampAlibabaMountEffortFor(schemas.Alibaba, "glm-5.3", "xhigh"))
+ for _, provider := range []schemas.ModelProvider{schemas.Zhipu, schemas.Kimi, schemas.Anthropic} {
+ assert.Equal(t, "xhigh", clampAlibabaMountEffortFor(provider, "glm-5.3", "xhigh"),
+ "provider %s must not clamp (verbatim rule preserved)", provider)
+ }
+}
+
+// TestAlibabaAnthropicMount_OpenAIDialectResponsesEffort covers the other
+// half of the alibaba mount: when an Anthropic-protocol request is served
+// through the OpenAI-compatible surface, the effort flows into a native
+// /responses call whose reasoning.effort runs through the shared OpenAI-dialect
+// ladder. qwen3.8-max tops out at xhigh there (vendor enum: none/minimal/low/
+// medium/high/xhigh), so xhigh forwards verbatim and max clamps down to xhigh —
+// unlike the Anthropic mount above, the gateway, not the vendor, does the
+// mapping.
+func TestAlibabaAnthropicMount_OpenAIDialectResponsesEffort(t *testing.T) {
+ t.Parallel()
+
+ for _, tc := range []struct{ effort, want string }{
+ {"xhigh", "xhigh"},
+ {"max", "xhigh"},
+ {"high", "high"},
+ } {
+ t.Run(tc.effort, func(t *testing.T) {
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("qwen3.8-max", schemas.Ptr(tc.effort), nil).ToBifrostResponsesRequest(ctx)
+ require.NotNil(t, bifrostReq)
+ bifrostReq.Provider = schemas.Alibaba
+
+ out := openai.ToOpenAIResponsesRequest(ctx, bifrostReq)
+ require.NotNil(t, out)
+ require.NotNil(t, out.ResponsesParameters.Reasoning, "reasoning was dropped on the alibaba responses path")
+ require.NotNil(t, out.ResponsesParameters.Reasoning.Effort)
+ assert.Equal(t, tc.want, *out.ResponsesParameters.Reasoning.Effort)
+ })
+ }
+}
+
+// TestAlibabaChatPath_EffortAloneNoSynthesizedThinking is the direct chat-with-
+// toggle regression (verified live 2026-08-23): a chat request carrying
+// reasoning_effort routed through use_anthropic_endpoints used to emit BOTH
+// thinking:{enabled,budget_tokens} (budget derived from the effort) and
+// output_config.effort, and the mount answered 400 ("'reasoning_effort' and
+// 'thinking_budget' cannot be set simultaneously"). The effort-only shape works
+// — the vendor engages thinking itself from the effort value — so the effort
+// travels alone. Covers the effort-only input and the co-present
+// reasoning.max_tokens + effort input (the mount's constraint is field-level,
+// not input-path-level), plus the max→xhigh clamp on the same path.
+func TestAlibabaChatPath_EffortAloneNoSynthesizedThinking(t *testing.T) {
+ t.Parallel()
+
+ for _, tc := range []struct {
+ name string
+ effort string
+ budget *int
+ wantEff string
+ }{
+ {name: "effort only, high", effort: "high", wantEff: "high"},
+ {name: "effort only, xhigh", effort: "xhigh", wantEff: "xhigh"},
+ {name: "effort only, max clamps to xhigh", effort: "max", wantEff: "xhigh"},
+ {name: "co-present reasoning.max_tokens, high", effort: "high", budget: schemas.Ptr(32000), wantEff: "high"},
+ {name: "co-present reasoning.max_tokens, max clamps to xhigh", effort: "max", budget: schemas.Ptr(32000), wantEff: "xhigh"},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ bifrostReq := &schemas.BifrostChatRequest{
+ Provider: schemas.Alibaba,
+ Model: "qwen3.8-max",
+ Input: []schemas.ChatMessage{
+ {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("What is 1+1?")}},
+ },
+ Params: &schemas.ChatParameters{
+ MaxCompletionTokens: schemas.Ptr(128000),
+ Reasoning: &schemas.ChatReasoning{Effort: &tc.effort, MaxTokens: tc.budget},
+ },
+ }
+
+ out, err := ToAnthropicChatRequest(providerEffortTestCtx(t), bifrostReq)
+ require.NoError(t, err)
+
+ require.NotNil(t, out.OutputConfig, "chat-path effort was dropped on the alibaba mount")
+ require.NotNil(t, out.OutputConfig.Effort)
+ assert.Equal(t, tc.wantEff, *out.OutputConfig.Effort)
+ assert.Nil(t, out.Thinking, "the alibaba mount rejects effort + thinking_budget together; the effort must travel alone")
+ })
+ }
+}
+
+// TestAlibabaAnthropicMount_NoEffortNoOutputConfig pins the existing no-effort
+// behavior: without an effort knob nothing is forwarded and nothing synthesized.
+func TestAlibabaAnthropicMount_NoEffortNoOutputConfig(t *testing.T) {
+ t.Parallel()
+
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("qwen3.8-max", nil, nil).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Alibaba
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ if out.OutputConfig != nil {
+ assert.Nil(t, out.OutputConfig.Effort, "no effort was requested; none may be forwarded")
+ }
+ assert.Nil(t, out.Thinking, "no reasoning signal was given; the model default applies")
+}
+
+// TestKimiAnthropicMount_EffortStripped pins the fail-closed behavior: Kimi's
+// /anthropic mount documents no effort equivalent (MoonshotAI/Kimi-K2#129), so
+// the field is stripped rather than forwarded as an unknown field. Thinking is
+// NOT synthesized to compensate — an Anthropic-protocol caller who omitted
+// thinking gets the model's own default (k2.6 defaults thinking on), matching
+// the don't-invent-thinking rule the Anthropic provider follows.
+func TestKimiAnthropicMount_EffortStripped(t *testing.T) {
+ t.Parallel()
+
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("kimi-k2.6", schemas.Ptr("max"), nil).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Kimi
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ if out.OutputConfig != nil {
+ assert.Nil(t, out.OutputConfig.Effort, "kimi's /anthropic mount has no effort equivalent; the field must not be forwarded")
+ }
+ assert.Nil(t, out.Thinking, "caller sent no thinking parameter; Bifrost must not invent one")
+}
+
+// TestZhipuAnthropicMount_DisabledThinkingRewritten covers the GLM-5.3 quirk:
+// z.ai 400s on thinking.type:"disabled" for forced-thinking models ("This model
+// always engages in thinking and cannot be disabled"), so it is rewritten to
+// enabled with the minimum budget — the closest legal shape to the caller's
+// "think as little as possible" intent.
+func TestZhipuAnthropicMount_DisabledThinkingRewritten(t *testing.T) {
+ t.Parallel()
+
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("glm-5.3", nil, &AnthropicThinking{Type: "disabled"}).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Zhipu
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ require.NotNil(t, out.Thinking, "GLM-5.3 rejects thinking.type:\"disabled\"; it must be rewritten, not forwarded")
+ assert.Equal(t, "enabled", out.Thinking.Type)
+ require.NotNil(t, out.Thinking.BudgetTokens)
+ assert.Equal(t, MinimumReasoningMaxTokens, *out.Thinking.BudgetTokens)
+}
+
+// TestZhipuAnthropicMount_DisabledThinkingPreservedOnGLM52 is the control:
+// GLM-5.2 supports disabling thinking, so the field passes through.
+func TestZhipuAnthropicMount_DisabledThinkingPreservedOnGLM52(t *testing.T) {
+ t.Parallel()
+
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("glm-5.2", nil, &AnthropicThinking{Type: "disabled"}).ToBifrostResponsesRequest(ctx)
+ bifrostReq.Provider = schemas.Zhipu
+
+ out := toAnthropicResponsesBuilt(t, ctx, bifrostReq)
+
+ require.NotNil(t, out.Thinking)
+ assert.Equal(t, "disabled", out.Thinking.Type)
+}
+
+// TestZhipuChatPath_EffortEmitsOutputConfig covers the OpenAI-dialect inbound
+// (reasoning_effort) on the zhipu Anthropic mount: effort lands on
+// output_config.effort and thinking is synthesized from it (the vendor requires
+// thinking enabled for effort to take effect).
+func TestZhipuChatPath_EffortEmitsOutputConfig(t *testing.T) {
+ t.Parallel()
+
+ effort := "max"
+ bifrostReq := &schemas.BifrostChatRequest{
+ Provider: schemas.Zhipu,
+ Model: "glm-5.3",
+ Input: []schemas.ChatMessage{
+ {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("What is 1+1?")}},
+ },
+ Params: &schemas.ChatParameters{
+ MaxCompletionTokens: schemas.Ptr(128000),
+ Reasoning: &schemas.ChatReasoning{Effort: &effort},
+ },
+ }
+
+ out, err := ToAnthropicChatRequest(providerEffortTestCtx(t), bifrostReq)
+ require.NoError(t, err)
+
+ require.NotNil(t, out.OutputConfig, "chat-path effort was dropped on the zhipu mount")
+ require.NotNil(t, out.OutputConfig.Effort)
+ assert.Equal(t, "max", *out.OutputConfig.Effort)
+ require.NotNil(t, out.Thinking)
+ assert.Equal(t, "enabled", out.Thinking.Type)
+ require.NotNil(t, out.Thinking.BudgetTokens)
+}
+
+// TestKimiChatPath_EffortStaysBudgetOnly is the chat-path control: kimi keeps
+// the budget-only shape.
+func TestKimiChatPath_EffortStaysBudgetOnly(t *testing.T) {
+ t.Parallel()
+
+ effort := "high"
+ bifrostReq := &schemas.BifrostChatRequest{
+ Provider: schemas.Kimi,
+ Model: "kimi-k2.6",
+ Input: []schemas.ChatMessage{
+ {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("What is 1+1?")}},
+ },
+ Params: &schemas.ChatParameters{
+ MaxCompletionTokens: schemas.Ptr(8192),
+ Reasoning: &schemas.ChatReasoning{Effort: &effort},
+ },
+ }
+
+ out, err := ToAnthropicChatRequest(providerEffortTestCtx(t), bifrostReq)
+ require.NoError(t, err)
+
+ if out.OutputConfig != nil {
+ assert.Nil(t, out.OutputConfig.Effort)
+ }
+ require.NotNil(t, out.Thinking)
+ assert.Equal(t, "enabled", out.Thinking.Type)
+}
+
+// TestZhipuChatPath_DisabledThinkingRewrittenOnGLM53 pins the chat-path half
+// of the forced-thinking rewrite (the chat converter strips internally).
+func TestZhipuChatPath_DisabledThinkingRewrittenOnGLM53(t *testing.T) {
+ t.Parallel()
+
+ none := "none"
+ bifrostReq := &schemas.BifrostChatRequest{
+ Provider: schemas.Zhipu,
+ Model: "glm-5.3",
+ Input: []schemas.ChatMessage{
+ {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("What is 1+1?")}},
+ },
+ Params: &schemas.ChatParameters{
+ MaxCompletionTokens: schemas.Ptr(8192),
+ Reasoning: &schemas.ChatReasoning{Effort: &none},
+ },
+ }
+
+ out, err := ToAnthropicChatRequest(providerEffortTestCtx(t), bifrostReq)
+ require.NoError(t, err)
+
+ require.NotNil(t, out.Thinking, "GLM-5.3 cannot disable thinking; it must be rewritten to enabled")
+ assert.Equal(t, "enabled", out.Thinking.Type)
+ require.NotNil(t, out.Thinking.BudgetTokens)
+ assert.Equal(t, MinimumReasoningMaxTokens, *out.Thinking.BudgetTokens)
+}
+
+// TestStripUnsupportedAnthropicFields_ProviderEffort covers the typed strip
+// gate: effort survives for the documented vendor families and is removed
+// everywhere else, with Anthropic's own model gate unchanged.
+func TestStripUnsupportedAnthropicFields_ProviderEffort(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ name string
+ provider schemas.ModelProvider
+ model string
+ kept bool
+ wantEffort string
+ }{
+ {"zhipu glm-5.3 keeps", schemas.Zhipu, "glm-5.3", true, "max"},
+ {"zhipu glm-5.2[1m] keeps", schemas.Zhipu, "glm-5.2[1m]", true, "max"},
+ {"zhipu glm-4.7 strips", schemas.Zhipu, "glm-4.7", false, ""},
+ {"alibaba qwen3.8 keeps", schemas.Alibaba, "qwen3.8-max", true, "xhigh"},
+ {"alibaba glm-5.2 keeps", schemas.Alibaba, "glm-5.2", true, "max"},
+ {"alibaba deepseek keeps", schemas.Alibaba, "deepseek-v4-pro", true, "max"},
+ {"alibaba qwen3.7 strips", schemas.Alibaba, "qwen3.7-plus", false, ""},
+ {"kimi strips", schemas.Kimi, "kimi-k3", false, ""},
+ {"anthropic haiku strips", schemas.Anthropic, "claude-haiku-4-5", false, ""},
+ {"anthropic opus keeps", schemas.Anthropic, "claude-opus-4-6", true, "max"},
+ }
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ req := &AnthropicMessageRequest{
+ Model: tc.model,
+ MaxTokens: 4096,
+ OutputConfig: &AnthropicOutputConfig{Effort: schemas.Ptr("max")},
+ }
+ stripUnsupportedAnthropicFields(req, tc.provider, tc.model)
+ if tc.kept {
+ require.NotNil(t, req.OutputConfig, "output_config was stripped for %s/%s", tc.provider, tc.model)
+ require.NotNil(t, req.OutputConfig.Effort)
+ // z.ai maps out-of-scale values server-side, so max stays
+ // verbatim there; Model Studio rejects out-of-enum values, so
+ // the strip pass clamps to each family's officially valid
+ // values for the alibaba profile.
+ assert.Equal(t, tc.wantEffort, *req.OutputConfig.Effort)
+ } else if req.OutputConfig != nil {
+ assert.Nil(t, req.OutputConfig.Effort)
+ }
+ })
+ }
+}
+
+// TestStripUnsupportedAnthropicFields_ZhipuForcedThinking covers the typed
+// thinking rewrite/synthesis for forced-thinking GLM models.
+func TestStripUnsupportedAnthropicFields_ZhipuForcedThinking(t *testing.T) {
+ t.Parallel()
+
+ // disabled → rewritten to enabled + minimum budget on GLM-5.3.
+ req := &AnthropicMessageRequest{
+ Model: "glm-5.3",
+ MaxTokens: 4096,
+ Thinking: &AnthropicThinking{Type: "disabled"},
+ }
+ stripUnsupportedAnthropicFields(req, schemas.Zhipu, "glm-5.3")
+ require.NotNil(t, req.Thinking, "thinking.type:\"disabled\" must be rewritten for GLM-5.3, not forwarded")
+ assert.Equal(t, "enabled", req.Thinking.Type)
+ require.NotNil(t, req.Thinking.BudgetTokens)
+ assert.Equal(t, MinimumReasoningMaxTokens, *req.Thinking.BudgetTokens)
+
+ // Same rewrite on GLM-4.7 (forced-thinking but tolerant of an absent field).
+ req47 := &AnthropicMessageRequest{
+ Model: "glm-4.7",
+ MaxTokens: 4096,
+ Thinking: &AnthropicThinking{Type: "disabled"},
+ }
+ stripUnsupportedAnthropicFields(req47, schemas.Zhipu, "glm-4.7")
+ require.NotNil(t, req47.Thinking)
+ assert.Equal(t, "enabled", req47.Thinking.Type)
+
+ // GLM-5.2 keeps disabled (control).
+ req52 := &AnthropicMessageRequest{
+ Model: "glm-5.2",
+ MaxTokens: 4096,
+ Thinking: &AnthropicThinking{Type: "disabled"},
+ }
+ stripUnsupportedAnthropicFields(req52, schemas.Zhipu, "glm-5.2")
+ require.NotNil(t, req52.Thinking, "GLM-5.2 accepts thinking.type:\"disabled\"")
+ assert.Equal(t, "disabled", req52.Thinking.Type)
+
+ // Absent thinking → synthesized on GLM-5.3, budget derived from effort.
+ reqSynth := &AnthropicMessageRequest{
+ Model: "glm-5.3",
+ MaxTokens: 128000,
+ OutputConfig: &AnthropicOutputConfig{Effort: schemas.Ptr("max")},
+ }
+ stripUnsupportedAnthropicFields(reqSynth, schemas.Zhipu, "glm-5.3")
+ require.NotNil(t, reqSynth.Thinking, "GLM-5.3 requires an explicit thinking field")
+ assert.Equal(t, "enabled", reqSynth.Thinking.Type)
+ require.NotNil(t, reqSynth.Thinking.BudgetTokens)
+ assert.Greater(t, *reqSynth.Thinking.BudgetTokens, MinimumReasoningMaxTokens, "effort=max should derive a large budget")
+
+ // Absent thinking stays absent on GLM-5.2 (control).
+ req52Absent := &AnthropicMessageRequest{Model: "glm-5.2", MaxTokens: 4096}
+ stripUnsupportedAnthropicFields(req52Absent, schemas.Zhipu, "glm-5.2")
+ assert.Nil(t, req52Absent.Thinking)
+}
+
+// TestStripUnsupportedFieldsFromRawBody_ProviderEffort is the raw-passthrough
+// counterpart: same provider/model matrix against the JSON body bytes.
+func TestStripUnsupportedFieldsFromRawBody_ProviderEffort(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ name string
+ provider schemas.ModelProvider
+ model string
+ kept bool
+ wantEffort string
+ }{
+ {"zhipu glm-5.3 keeps", schemas.Zhipu, "glm-5.3", true, "max"},
+ {"alibaba qwen3.8 keeps", schemas.Alibaba, "qwen3.8-max", true, "xhigh"},
+ {"alibaba glm-5.2 keeps max", schemas.Alibaba, "glm-5.2", true, "max"},
+ {"alibaba prefixed glm-5.3 keeps max", schemas.Alibaba, "ZHIPU/GLM-5.3", true, "max"},
+ {"alibaba qwen3-max strips", schemas.Alibaba, "qwen3-max", false, ""},
+ {"kimi strips", schemas.Kimi, "kimi-k2.6", false, ""},
+ }
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ body := []byte(`{"model":"` + tc.model + `","max_tokens":4096,"messages":[{"role":"user","content":"hi"}],"output_config":{"effort":"max"}}`)
+ out, err := StripUnsupportedFieldsFromRawBody(body, tc.provider, tc.model)
+ require.NoError(t, err)
+ got := providerUtils.GetJSONField(out, "output_config.effort")
+ if tc.kept {
+ require.True(t, got.Exists(), "raw output_config.effort was stripped for %s/%s", tc.provider, tc.model)
+ assert.Equal(t, tc.wantEffort, got.String())
+ } else {
+ assert.False(t, got.Exists(), "raw output_config.effort must be stripped for %s/%s", tc.provider, tc.model)
+ }
+ })
+ }
+}
+
+// TestResolveAnthropicMountProfile pins the host→profile mapping used for
+// custom providers (base_provider_type: anthropic). The reported bug's live
+// configuration was exactly this: a custom provider named "zai-anthropic"
+// pointed at https://api.z.ai/api/anthropic, which must get Zhipu's capability
+// profile, not Anthropic's model gates.
+func TestResolveAnthropicMountProfile(t *testing.T) {
+ t.Parallel()
+
+ cases := []struct {
+ baseURL string
+ want schemas.ModelProvider
+ }{
+ {"https://api.z.ai/api/anthropic", schemas.Zhipu},
+ {"https://api.z.ai/api/coding/paas/v4", schemas.Zhipu},
+ {"https://open.bigmodel.cn/api/anthropic", schemas.Zhipu},
+ {"https://dashscope-intl.aliyuncs.com/apps/anthropic", schemas.Alibaba},
+ {"https://ws-abc123.ap-southeast-1.maas.aliyuncs.com/apps/anthropic", schemas.Alibaba},
+ {"https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic", schemas.Alibaba},
+ {"https://api.moonshot.ai/anthropic", schemas.Kimi},
+ {"https://api.moonshot.cn/anthropic", schemas.Kimi},
+ {"https://api.kimi.com/coding", schemas.Kimi},
+ {"https://api.anthropic.com", schemas.Anthropic},
+ {"https://my-proxy.internal/anthropic", schemas.Anthropic},
+ {"", schemas.Anthropic},
+ }
+ for _, tc := range cases {
+ t.Run(tc.baseURL, func(t *testing.T) {
+ assert.Equal(t, tc.want, ResolveAnthropicMountProfile(tc.baseURL))
+ })
+ }
+}
+
+// TestAnthropicProvider_ConversionProviderAndNormalization covers the custom
+// provider wiring: the build config and the request's Provider both resolve to
+// the mount profile, while the stock provider and custom-anthropic providers
+// are untouched.
+func TestAnthropicProvider_ConversionProviderAndNormalization(t *testing.T) {
+ t.Parallel()
+
+ custom := &AnthropicProvider{
+ customProviderConfig: &schemas.CustomProviderConfig{BaseProviderType: schemas.Anthropic},
+ networkConfig: schemas.NetworkConfig{BaseURL: "https://api.z.ai/api/anthropic"},
+ }
+ assert.Equal(t, schemas.Zhipu, custom.conversionProvider())
+
+ // Request carrying the custom provider's own name is re-tagged to the
+ // resolved profile (shallow copy — the caller's request is never mutated).
+ req := &schemas.BifrostChatRequest{Provider: schemas.ModelProvider("zai-anthropic"), Model: "glm-5.3"}
+ normalized := custom.normalizeChatRequestForConversion(req)
+ assert.Equal(t, schemas.Zhipu, normalized.Provider)
+ assert.Equal(t, schemas.ModelProvider("zai-anthropic"), req.Provider, "caller's request must not be mutated")
+
+ respReq := &schemas.BifrostResponsesRequest{Provider: schemas.ModelProvider("zai-anthropic"), Model: "glm-5.3"}
+ normalizedResp := custom.normalizeResponsesRequestForConversion(respReq)
+ assert.Equal(t, schemas.Zhipu, normalizedResp.Provider)
+
+ // Stock provider: no normalization, profile stays Anthropic.
+ stock := &AnthropicProvider{}
+ assert.Equal(t, schemas.Anthropic, stock.conversionProvider())
+ assert.Same(t, req, stock.normalizeChatRequestForConversion(req))
+
+ // Custom provider pointed at Anthropic itself: no normalization (existing
+ // behavior preserved bit-for-bit).
+ customAnthropic := &AnthropicProvider{
+ customProviderConfig: &schemas.CustomProviderConfig{BaseProviderType: schemas.Anthropic},
+ networkConfig: schemas.NetworkConfig{BaseURL: "https://api.anthropic.com"},
+ }
+ assert.Equal(t, schemas.Anthropic, customAnthropic.conversionProvider())
+ assert.Same(t, req, customAnthropic.normalizeChatRequestForConversion(req))
+
+ // Kimi-host custom provider resolves to the Kimi profile.
+ customKimi := &AnthropicProvider{
+ customProviderConfig: &schemas.CustomProviderConfig{BaseProviderType: schemas.Anthropic},
+ networkConfig: schemas.NetworkConfig{BaseURL: "https://api.kimi.com/coding"},
+ }
+ assert.Equal(t, schemas.Kimi, customKimi.conversionProvider())
+}
+
+// TestCustomProviderZaiMount_EffortAndThinkingRoundTrip is the end-to-end pin
+// for the reported configuration: a request normalized through a
+// "zai-anthropic"-style custom provider profile emits effort + thinking exactly
+// like the built-in zhipu mount.
+func TestCustomProviderZaiMount_EffortAndThinkingRoundTrip(t *testing.T) {
+ t.Parallel()
+
+ custom := &AnthropicProvider{
+ customProviderConfig: &schemas.CustomProviderConfig{BaseProviderType: schemas.Anthropic},
+ networkConfig: schemas.NetworkConfig{BaseURL: "https://api.z.ai/api/anthropic"},
+ }
+
+ ctx := providerEffortTestCtx(t)
+ bifrostReq := anthropicInbound("glm-5.3", schemas.Ptr("max"), nil).ToBifrostResponsesRequest(ctx)
+ require.NotNil(t, bifrostReq)
+ bifrostReq.Provider = schemas.ModelProvider("zai-anthropic") // what routing assigns
+
+ normalized := custom.normalizeResponsesRequestForConversion(bifrostReq)
+ out := toAnthropicResponsesBuilt(t, ctx, normalized)
+
+ require.NotNil(t, out.OutputConfig, "output_config was dropped on the custom z.ai mount")
+ require.NotNil(t, out.OutputConfig.Effort)
+ assert.Equal(t, "max", *out.OutputConfig.Effort)
+ require.NotNil(t, out.Thinking, "GLM-5.3 requires an explicit thinking field on z.ai's mount")
+ assert.Equal(t, "enabled", out.Thinking.Type)
+}
+
+// TestStripUnsupportedFieldsFromRawBody_ZhipuForcedThinking covers the raw-path
+// thinking rewrite/synthesis for GLM-5.3, and its GLM-5.2 control.
+func TestStripUnsupportedFieldsFromRawBody_ZhipuForcedThinking(t *testing.T) {
+ t.Parallel()
+
+ // disabled → rewritten to enabled + minimum budget.
+ body53 := []byte(`{"model":"glm-5.3","max_tokens":4096,"messages":[{"role":"user","content":"hi"}],"thinking":{"type":"disabled"}}`)
+ out, err := StripUnsupportedFieldsFromRawBody(body53, schemas.Zhipu, "glm-5.3")
+ require.NoError(t, err)
+ assert.Equal(t, "enabled", providerUtils.GetJSONField(out, "thinking.type").String(), "thinking must be rewritten for GLM-5.3 on the raw path")
+ assert.Equal(t, int64(MinimumReasoningMaxTokens), providerUtils.GetJSONField(out, "thinking.budget_tokens").Int())
+
+ // Absent thinking → synthesized; budget derived from output_config.effort.
+ bodySynth := []byte(`{"model":"glm-5.3","max_tokens":128000,"messages":[{"role":"user","content":"hi"}],"output_config":{"effort":"max"}}`)
+ outSynth, err := StripUnsupportedFieldsFromRawBody(bodySynth, schemas.Zhipu, "glm-5.3")
+ require.NoError(t, err)
+ assert.Equal(t, "enabled", providerUtils.GetJSONField(outSynth, "thinking.type").String(), "GLM-5.3 requires an explicit thinking field")
+ assert.Greater(t, providerUtils.GetJSONField(outSynth, "thinking.budget_tokens").Int(), int64(MinimumReasoningMaxTokens), "effort=max should derive a large budget")
+ assert.Equal(t, "max", providerUtils.GetJSONField(outSynth, "output_config.effort").String(), "effort survives alongside the synthesized thinking")
+
+ // GLM-5.2 keeps disabled (control).
+ body52 := []byte(`{"model":"glm-5.2","max_tokens":4096,"messages":[{"role":"user","content":"hi"}],"thinking":{"type":"disabled"}}`)
+ out52, err := StripUnsupportedFieldsFromRawBody(body52, schemas.Zhipu, "glm-5.2")
+ require.NoError(t, err)
+ assert.Equal(t, "disabled", providerUtils.GetJSONField(out52, "thinking.type").String(), "GLM-5.2 accepts thinking.type:\"disabled\"")
+
+ // GLM-5.2 absent thinking stays absent (control).
+ body52Absent := []byte(`{"model":"glm-5.2","max_tokens":4096,"messages":[{"role":"user","content":"hi"}]}`)
+ out52Absent, err := StripUnsupportedFieldsFromRawBody(body52Absent, schemas.Zhipu, "glm-5.2")
+ require.NoError(t, err)
+ assert.False(t, providerUtils.JSONFieldExists(out52Absent, "thinking"))
+}
diff --git a/core/providers/anthropic/requestbuilder.go b/core/providers/anthropic/requestbuilder.go
index 6c2e46fbe87..7797098d0ad 100644
--- a/core/providers/anthropic/requestbuilder.go
+++ b/core/providers/anthropic/requestbuilder.go
@@ -126,6 +126,9 @@ var AnthropicProviderRequestDefaultsMap = map[schemas.ModelProvider]AnthropicPro
InlineURLSources: true,
},
schemas.DeepSeek: {},
+ schemas.Alibaba: {},
+ schemas.Kimi: {},
+ schemas.Zhipu: {},
// Vertex publisher endpoint: model + region in URL, anthropic_version
// required, beta headers in body (not HTTP), cache_control.scope stripped
// at marshal time, tool versions remapped.
diff --git a/core/providers/anthropic/responses.go b/core/providers/anthropic/responses.go
index 960d1942f92..d4ce0d1a67e 100644
--- a/core/providers/anthropic/responses.go
+++ b/core/providers/anthropic/responses.go
@@ -4188,7 +4188,7 @@ func ToAnthropicResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schema
// Preserve a co-present effort — these models support
// output_config.effort, and the budget is otherwise dropped.
if bifrostReq.Params.Reasoning.Effort != nil && *bifrostReq.Params.Reasoning.Effort != "none" {
- setEffortOnOutputConfig(anthropicReq, MapBifrostEffortToAnthropic(*bifrostReq.Params.Reasoning.Effort))
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, MapBifrostEffortToAnthropic(*bifrostReq.Params.Reasoning.Effort))
}
} else {
budgetTokens := *bifrostReq.Params.Reasoning.MaxTokens
@@ -4204,15 +4204,32 @@ func ToAnthropicResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schema
Type: "enabled",
BudgetTokens: schemas.Ptr(budgetTokens),
}
+ // Vendor extension mounts keep a co-present effort instead
+ // of discarding it. z.ai accepts thinking.budget_tokens and
+ // output_config.effort together (the ZCode-proven shape);
+ // Model Studio rejects the pair ("'reasoning_effort' and
+ // 'thinking_budget' cannot be set simultaneously") and
+ // engages thinking itself from the effort value, so the
+ // effort wins there and the thinking field is dropped
+ // (verified live 2026-08-23).
+ if bifrostReq.Params.Reasoning.Effort != nil && *bifrostReq.Params.Reasoning.Effort != "none" &&
+ SupportsProviderEffort(bifrostReq.Provider, capModel) {
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, MapBifrostEffortToAnthropic(*bifrostReq.Params.Reasoning.Effort))
+ if bifrostReq.Provider == schemas.Alibaba {
+ anthropicReq.Thinking = nil
+ }
+ }
}
} else if native, ok := anthropicNativeEffortFrom(ctx); ok && native.ThinkingOmitted {
// The caller sent output_config.effort and no thinking parameter.
// Forward exactly that: the effort alone, with thinking left absent so
// the model applies its own default. Synthesizing thinking here would
// turn it on against the caller's request on every model that defaults
- // it off (Opus 4.6/4.7/4.8, Sonnet 4.6).
- if caps.SupportsNativeEffort(DefaultSupportsNativeEffort(caps.Model())) {
- setEffortOnOutputConfig(anthropicReq, MapBifrostEffortToAnthropic(native.Effort))
+ // it off (Opus 4.6/4.7/4.8, Sonnet 4.6). Provider-aware via
+ // SupportsProviderEffort: vendor mounts (z.ai, Model Studio) keep the
+ // field for their documented families too.
+ if SupportsProviderEffort(bifrostReq.Provider, capModel) {
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, MapBifrostEffortToAnthropic(native.Effort))
}
} else {
if bifrostReq.Params.Reasoning.Effort != nil {
@@ -4222,17 +4239,24 @@ func ToAnthropicResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schema
if caps.SupportsAdaptiveThinking(DefaultSupportsAdaptiveThinking(caps.Model())) {
// Opus 4.6+ and Opus 4.7+: adaptive thinking + native effort
anthropicReq.Thinking = &AnthropicThinking{Type: "adaptive"}
- setEffortOnOutputConfig(anthropicReq, effort)
- } else if SupportsNativeEffort(caps) {
- // Opus 4.5: native effort + budget_tokens thinking
- setEffortOnOutputConfig(anthropicReq, effort)
- budgetTokens, err := providerUtils.GetBudgetTokensFromReasoningEffort(effort, MinimumReasoningMaxTokens, anthropicReq.MaxTokens)
- if err != nil {
- return nil, fmt.Errorf("%w: %w", ErrReasoningMaxTokensTooLow, err)
- }
- anthropicReq.Thinking = &AnthropicThinking{
- Type: "enabled",
- BudgetTokens: schemas.Ptr(budgetTokens),
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, effort)
+ } else if SupportsNativeEffort(caps) || SupportsProviderEffort(bifrostReq.Provider, capModel) {
+ // Opus 4.5: native effort + budget_tokens thinking.
+ // z.ai (GLM-5.2+) takes the same shape — the mount maps
+ // the effort value server-side. Model Studio (Alibaba)
+ // rejects effort + thinking_budget together and engages
+ // thinking itself from the effort value, so the effort is
+ // forwarded alone there (verified live 2026-08-23).
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, effort)
+ if bifrostReq.Provider != schemas.Alibaba {
+ budgetTokens, err := providerUtils.GetBudgetTokensFromReasoningEffort(effort, MinimumReasoningMaxTokens, anthropicReq.MaxTokens)
+ if err != nil {
+ return nil, fmt.Errorf("%w: %w", ErrReasoningMaxTokensTooLow, err)
+ }
+ anthropicReq.Thinking = &AnthropicThinking{
+ Type: "enabled",
+ BudgetTokens: schemas.Ptr(budgetTokens),
+ }
}
} else {
// Older models: budget_tokens only
@@ -4250,7 +4274,9 @@ func ToAnthropicResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schema
// Fable/Mythos reject thinking:{type:"disabled"} with a 400 —
// adaptive thinking is always on and cannot be disabled. Omit
// the thinking param entirely for that family; all other
- // models take the explicit disabled path.
+ // models take the explicit disabled path. (Forced-thinking
+ // GLM models get rewritten to enabled downstream in
+ // stripUnsupportedAnthropicFields.)
anthropicReq.Thinking = &AnthropicThinking{
Type: "disabled",
}
@@ -4261,8 +4287,8 @@ func ToAnthropicResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schema
// The neutral params collapsed the caller's effort into "none"
// to signal reasoning-off, so restore it from what they sent.
if native, ok := anthropicNativeEffortFrom(ctx); ok && native.Effort != "" &&
- caps.SupportsNativeEffort(DefaultSupportsNativeEffort(caps.Model())) {
- setEffortOnOutputConfig(anthropicReq, MapBifrostEffortToAnthropic(native.Effort))
+ SupportsProviderEffort(bifrostReq.Provider, capModel) {
+ setEffortOnOutputConfig(anthropicReq, bifrostReq.Provider, capModel, MapBifrostEffortToAnthropic(native.Effort))
}
}
}
@@ -4966,8 +4992,45 @@ func ConvertBifrostMessagesToAnthropicMessages(ctx *schemas.BifrostContext, bifr
if i == len(bifrostMessages)-1 {
return true
}
- next := bifrostMessages[i+1]
- return next.Role != nil && *next.Role == schemas.ResponsesInputMessageRoleAssistant
+ // The "followed by an assistant turn" clause is evaluated against the next
+ // ROLE-BEARING message. Roleless reasoning and function_call items are
+ // intermediate representations of the adjacent assistant turn — Anthropic's
+ // own wire carries thinking and tool_use INSIDE the assistant message, and
+ // the egress below regroups these fragments into it — so they are skipped.
+ // Failing on them was a false negative that forced the fallback (hoist or
+ // inline) for exactly the thinking-replay shape every extended-thinking
+ // conversation produces on replay. A function_call fragment confirms the
+ // regrouped assistant message is emitted directly after the system turn
+ // (its tool results follow it, never precede it), which settles the clause
+ // then and there. A roleless function_call_output with no function_call
+ // ahead of it is the USER side of a tool exchange (it regroups into a user
+ // turn), so it and any other roleless shape fail the clause instead.
+ // Known theoretical false positive, unreachable from opencode-shaped
+ // traffic (which anchors system entries right after a role-bearing user
+ // message): a hand-crafted [user, function_call_output, system,
+ // function_call] sequence — clause 1 passes (the buffered output flushes
+ // later), the scan sees function_call and returns true, but the egress
+ // flushes the buffered results as a USER message between the system turn
+ // and the regrouped assistant, yielding [user, system, user, assistant].
+ for j := i + 1; j < len(bifrostMessages); j++ {
+ next := bifrostMessages[j]
+ if next.Role != nil {
+ return *next.Role == schemas.ResponsesInputMessageRoleAssistant
+ }
+ if next.Type == nil {
+ return false
+ }
+ switch *next.Type {
+ case schemas.ResponsesMessageTypeFunctionCall:
+ return true
+ case schemas.ResponsesMessageTypeReasoning:
+ continue
+ }
+ return false
+ }
+ // Only reasoning fragments followed the system turn; they regroup into
+ // an assistant message, satisfying the clause.
+ return true
}
// Helper to emit orphaned tool results (no matching tool_use) as a single user
diff --git a/core/providers/anthropic/roundtrip_test.go b/core/providers/anthropic/roundtrip_test.go
index 6c8d4f664ff..dcba5d1ffd0 100644
--- a/core/providers/anthropic/roundtrip_test.go
+++ b/core/providers/anthropic/roundtrip_test.go
@@ -11,11 +11,15 @@ package anthropic
// F) No top-level system, mid-conv only, unsupported
// G) Multiple mid-conv system messages, supported
// H) Multiple mid-conv system messages, unsupported
+// B3) Mid-conv system followed by a thinking-replay assistant turn, supported
+// B4) Mid-conv system followed by a textless tool_use assistant turn, supported
+// B5) Mid-conv system followed by a tool-result user turn — NOT forwarded
//
// "Supported" means provider=Anthropic + model=claude-opus-4-8 (SupportsMidConversationSystem=true).
import (
"context"
+ "encoding/json"
"strings"
"testing"
@@ -518,3 +522,118 @@ func TestRoundTrip_ContainerUpload_Grouped(t *testing.T) {
t.Errorf("file_id = %v, want %q", found.FileID, fileID)
}
}
+
+// --- B3: mid-conv system followed by a thinking-replay assistant turn -------
+
+// The replay of an extended-thinking conversation ingests each thinking block
+// as a ROLELESS reasoning item ahead of the assistant message's own item. On
+// Anthropic's wire the thinking rides INSIDE the assistant message, so the
+// emitted shape is still [user, system, assistant] and both placement clauses
+// hold. The gate must evaluate "followed by an assistant turn" against the
+// next role-bearing message, skipping the assistant-side fragments — failing
+// on them forced a hoist/inline fallback for every thinking conversation.
+func TestRoundTrip_B3_MidConvFollowedByThinkingReplay_Supported(t *testing.T) {
+ thinking := "internal deliberation"
+ signature := "sig-1"
+ text := "Understood."
+ messages := []AnthropicMessage{
+ anthMsg(AnthropicMessageRoleUser, "Hello"),
+ anthMsg(AnthropicMessageRoleSystem, "From now on, be concise."),
+ {
+ Role: AnthropicMessageRoleAssistant,
+ Content: AnthropicContent{ContentBlocks: []AnthropicContentBlock{
+ {Type: AnthropicContentBlockTypeThinking, Thinking: &thinking, Signature: &signature},
+ {Type: AnthropicContentBlockTypeText, Text: &text},
+ }},
+ },
+ anthMsg(AnthropicMessageRoleUser, "Thanks."),
+ }
+ system := systemStr("You are a helpful assistant.")
+
+ outMsgs, outSystem := roundTrip(t, messages, system, schemas.Anthropic, "claude-opus-4-8")
+
+ if got := textBlocks(outSystem); len(got) != 1 || got[0] != "You are a helpful assistant." {
+ t.Errorf("system = %v, want only the top-level system (no hoist)", got)
+ }
+ if want := "user,system,assistant,user"; roleSeq(outMsgs) != want {
+ t.Fatalf("role seq = %q, want %q", roleSeq(outMsgs), want)
+ }
+ if got := textBlocks(&outMsgs[1].Content); len(got) == 0 || got[0] != "From now on, be concise." {
+ t.Errorf("mid-conv system text = %v", got)
+ }
+}
+
+// --- B4: mid-conv system followed by a textless tool_use assistant turn -----
+
+// An assistant turn carrying only thinking + tool_use produces NO role-bearing
+// item of its own (reasoning and function_call fragments only). The gate must
+// still forward the system turn natively — the egress regroups the fragments
+// into the assistant message that satisfies the clause.
+func TestRoundTrip_B4_MidConvFollowedByToolUseOnly_Supported(t *testing.T) {
+ thinking := "deciding to call"
+ signature := "sig-2"
+ callID := "toolu_1"
+ toolName := "get_weather"
+ input := json.RawMessage(`{"city":"Paris"}`)
+ result := `{"temp":21}`
+ messages := []AnthropicMessage{
+ anthMsg(AnthropicMessageRoleUser, "Weather in Paris?"),
+ anthMsg(AnthropicMessageRoleSystem, "From now on, be concise."),
+ {
+ Role: AnthropicMessageRoleAssistant,
+ Content: AnthropicContent{ContentBlocks: []AnthropicContentBlock{
+ {Type: AnthropicContentBlockTypeThinking, Thinking: &thinking, Signature: &signature},
+ {Type: AnthropicContentBlockTypeToolUse, ID: &callID, Name: &toolName, Input: input},
+ }},
+ },
+ {
+ Role: AnthropicMessageRoleUser,
+ Content: AnthropicContent{ContentBlocks: []AnthropicContentBlock{
+ {Type: AnthropicContentBlockTypeToolResult, ToolUseID: &callID, Content: &AnthropicContent{ContentStr: &result}},
+ }},
+ },
+ }
+ system := systemStr("You are a helpful assistant.")
+
+ outMsgs, outSystem := roundTrip(t, messages, system, schemas.Anthropic, "claude-opus-4-8")
+
+ if got := textBlocks(outSystem); len(got) != 1 || got[0] != "You are a helpful assistant." {
+ t.Errorf("system = %v, want only the top-level system (no hoist)", got)
+ }
+ if want := "user,system,assistant,user"; roleSeq(outMsgs) != want {
+ t.Fatalf("role seq = %q, want %q", roleSeq(outMsgs), want)
+ }
+}
+
+// --- B5: mid-conv system followed by a tool-result user turn (still fails) --
+
+// A roleless function_call_output is the USER side of a tool exchange: it
+// regroups into a user turn, so a system turn followed by one must NOT be
+// forwarded natively ([user, system, user] is a 400 on Anthropic). The skip
+// in midConvPlacementOK covers assistant-side fragments only.
+func TestRoundTrip_B5_MidConvFollowedByToolOutput_NotForwarded(t *testing.T) {
+ callID := "toolu_2"
+ result := "21"
+ messages := []AnthropicMessage{
+ {
+ Role: AnthropicMessageRoleUser,
+ Content: AnthropicContent{ContentBlocks: []AnthropicContentBlock{
+ {Type: AnthropicContentBlockTypeToolResult, ToolUseID: &callID, Content: &AnthropicContent{ContentStr: &result}},
+ }},
+ },
+ anthMsg(AnthropicMessageRoleSystem, "From now on, be concise."),
+ anthMsg(AnthropicMessageRoleUser, "Thanks."),
+ }
+ system := systemStr("You are a helpful assistant.")
+
+ outMsgs, outSystem := roundTrip(t, messages, system, schemas.Anthropic, "claude-opus-4-8")
+
+ for i, m := range outMsgs {
+ if m.Role == AnthropicMessageRoleSystem {
+ t.Fatalf("msg[%d] forwarded as role:system; role seq = %q", i, roleSeq(outMsgs))
+ }
+ }
+ if got := textBlocks(outSystem); len(got) != 1 {
+ t.Logf("system blocks = %d (fallback applied, not native)", len(got))
+ }
+}
diff --git a/core/providers/anthropic/types.go b/core/providers/anthropic/types.go
index 5848580d5bc..9c54aa2f822 100644
--- a/core/providers/anthropic/types.go
+++ b/core/providers/anthropic/types.go
@@ -189,6 +189,19 @@ type ProviderFeatureSupport struct {
ServerSideFallback bool // native "fallbacks" request field — server-side-fallback-2026-06-01. Claude API only per docs ("not available on Amazon Bedrock, Google Cloud, or Microsoft Foundry").
FallbackCredit bool // fallback_credit_token request field + stop_details credit fields — fallback-credit-2026-06-01 (AWS surfaces: -2026-06-09). Documented on the Claude API, Amazon Bedrock, Google Cloud and Microsoft Foundry, i.e. the inverse of ServerSideFallback.
MidConvToolChanges bool // tool_addition/tool_removal blocks — mid-conversation-tool-changes-2026-07-01. Native Anthropic surface (Claude API + Bedrock Mantle); Bedrock is Opus 5 only, enforced upstream.
+ // OutputConfigEffort marks non-Anthropic Anthropic-compatible mounts that honor
+ // output_config.effort as an extension field. Anthropic's own
+ // support stays model-gated via SupportsEffortParameter; this flag covers vendors
+ // whose mounts accept (and, where documented, server-side map) the field for
+ // specific model families — see providerSupportsEffortModel for the per-vendor
+ // model scope.
+ // Cites: Z = https://docs.z.ai/guides/capabilities/thinking (GLM-5.2+ — documented
+ // there); Q = https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages
+ // (mount shape). Note Q documents NO effort field on the Anthropic mount — Model
+ // Studio honors output_config.effort as an undocumented passthrough to its
+ // OpenAI-dialect backend, whose per-model reasoning_effort ladder is documented
+ // on the Model Studio model page (live-verified 2026-08-23).
+ OutputConfigEffort bool
}
// ProviderFeatures maps each provider to its supported Anthropic features.
@@ -343,6 +356,37 @@ var ProviderFeatures = map[schemas.ModelProvider]ProviderFeatureSupport{
InterleavedThinking: true,
ServiceTier: true,
},
+ // Alibaba Cloud Model Studio / Kimi / Zhipu Anthropic-compatible mounts:
+ // start fail-closed (all Anthropic server-side features off). Enable flags
+ // only as vendor docs verify them (docs/research 02 §8.4, 03 §6.3, 04 §7.4).
+ //
+ // Zhipu — z.ai's Anthropic mount (Coding Plan only) accepts output_config.effort
+ // for GLM-5.2 and above; out-of-scale values are mapped server-side per
+ // https://docs.z.ai/guides/capabilities/thinking (verified 2026-08-16; live ZCode
+ // traffic shows thinking{budget_tokens} + output_config{effort} coexisting).
+ // InterleavedThinking is on because z.ai accepts the interleaved-thinking beta
+ // header Claude Code sends against this mount (docs/research 04 §7.4) — it only
+ // gates header passthrough, GLM serves interleaved thinking natively.
+ schemas.Zhipu: {
+ OutputConfigEffort: true,
+ InterleavedThinking: true,
+ },
+ // Alibaba — Model Studio's /apps/anthropic mount documents no effort field
+ // (its API page omits output_config.effort entirely); the field is nonetheless
+ // honored live: the mount proxies Messages to its OpenAI-dialect backend, which
+ // validates output_config.effort against the per-model reasoning_effort ladder
+ // documented on the Model Studio model page — it does NOT map out-of-enum values
+ // server-side (max 400s). Bifrost forwards the field with the per-model clamp in
+ // clampAlibabaMountEffortForModel; the model scope is enforced in
+ // providerSupportsEffortModel. Empirical contract, live-verified 2026-08-23.
+ schemas.Alibaba: {
+ OutputConfigEffort: true,
+ },
+ // Kimi — the /anthropic mount is still an unofficial, empirical contract
+ // (MoonshotAI/Kimi-K2#129): model, messages, system, tools, tool_choice,
+ // max_tokens, temperature, stream. No effort equivalent is documented, so
+ // output_config.effort stays stripped (K3's reasoning_effort is OpenAI-mount only).
+ schemas.Kimi: {},
}
// ==================== REQUEST TYPES ====================
diff --git a/core/providers/anthropic/utils.go b/core/providers/anthropic/utils.go
index 0e80b8afc2c..6c036a1c8d9 100644
--- a/core/providers/anthropic/utils.go
+++ b/core/providers/anthropic/utils.go
@@ -8,6 +8,7 @@ import (
"fmt"
"slices"
"sort"
+ "strconv"
"strings"
"github.com/bytedance/sonic"
@@ -250,13 +251,23 @@ func stripUnsupportedAnthropicFields(req *AnthropicMessageRequest, provider sche
// output_config.effort — model-gated per
// https://platform.claude.com/docs/en/build-with-claude/effort. Models
// outside the supported set return: "This model does not support the
- // effort parameter."
- if req.OutputConfig != nil && req.OutputConfig.Effort != nil && !caps.SupportsNativeEffort(DefaultSupportsNativeEffort(caps.Model())) {
+ // effort parameter." Provider-aware via SupportsProviderEffort: mounts that
+ // document the field as a vendor extension (z.ai, Model Studio) keep it for
+ // their documented families, and the fallback carries the datasheet-driven
+ // native-effort gate for everything else.
+ if req.OutputConfig != nil && req.OutputConfig.Effort != nil && !SupportsProviderEffort(provider, model) {
req.OutputConfig.Effort = nil
if req.OutputConfig.Format == nil && req.OutputConfig.TaskBudget == nil {
req.OutputConfig = nil
}
}
+ // Model Studio clamp: the alibaba mount validates a surviving effort value
+ // against its per-model chat-completions enum instead of mapping it
+ // server-side (see clampAlibabaMountEffortForModel), so each family's
+ // officially valid values are emitted on the typed path too.
+ if provider == schemas.Alibaba && req.OutputConfig != nil && req.OutputConfig.Effort != nil {
+ req.OutputConfig.Effort = schemas.Ptr(clampAlibabaMountEffortForModel(model, *req.OutputConfig.Effort))
+ }
// thinking.type — model-gated. Adaptive-only models (Opus 4.7+, Sonnet 5+,
// Fable/Mythos) removed extended thinking and reject the legacy shape with:
//
@@ -302,6 +313,31 @@ func stripUnsupportedAnthropicFields(req *AnthropicMessageRequest, provider sche
req.Thinking.BudgetTokens = nil
}
}
+ // Forced-thinking GLM models on z.ai's mount (GLM-5.3+, GLM-4.7, GLM-4.5V):
+ // thinking cannot be disabled — GLM-5.3+ 400s ("This model always engages in
+ // thinking and cannot be disabled") and the mount treats an absent thinking
+ // field as disabled. Rewrite disabled → enabled with the minimum budget (the
+ // closest legal shape to the caller's "cheap" intent).
+ if provider == schemas.Zhipu && req.Thinking != nil && req.Thinking.Type == "disabled" &&
+ ZhipuForcedThinkingModel(model) {
+ req.Thinking = &AnthropicThinking{
+ Type: "enabled",
+ BudgetTokens: schemas.Ptr(MinimumReasoningMaxTokens),
+ }
+ }
+ // GLM-5.3+ requires the thinking field outright on this mount (absent =
+ // disabled = 1210 error), so synthesize it. Budget precedence: effort-derived
+ // (the caller's output_config.effort) > minimum.
+ if provider == schemas.Zhipu && req.Thinking == nil && ZhipuRequiresThinkingModel(model) {
+ var effort *string
+ if req.OutputConfig != nil {
+ effort = req.OutputConfig.Effort
+ }
+ req.Thinking = &AnthropicThinking{
+ Type: "enabled",
+ BudgetTokens: schemas.Ptr(zhipuThinkingBudget(effort, req.MaxTokens)),
+ }
+ }
if req.InferenceGeo != nil && !caps.SupportsInferenceGeo(features.InferenceGeo) {
req.InferenceGeo = nil
}
@@ -619,9 +655,10 @@ func StripUnsupportedFieldsFromRawBody(jsonBody []byte, provider schemas.ModelPr
// output_config.effort — model-gated per
// https://platform.claude.com/docs/en/build-with-claude/effort.
- // Mirrors the typed path; same cleanup of an empty parent.
+ // Mirrors the typed path; same cleanup of an empty parent. Provider-aware
+ // via SupportsProviderEffort (z.ai / Model Studio extension families).
if providerUtils.JSONFieldExists(jsonBody, "output_config.effort") &&
- !caps.SupportsNativeEffort(DefaultSupportsNativeEffort(caps.Model())) {
+ !SupportsProviderEffort(provider, model) {
jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "output_config.effort")
if err != nil {
return nil, fmt.Errorf("strip raw output_config.effort: %w", err)
@@ -634,6 +671,20 @@ func StripUnsupportedFieldsFromRawBody(jsonBody []byte, provider schemas.ModelPr
}
}
+ // Model Studio clamp — mirrors the typed path: a surviving effort value is
+ // mapped onto the mount's per-model chat-completions enum instead of
+ // being forwarded into a vendor 400.
+ if provider == schemas.Alibaba {
+ if e := providerUtils.GetJSONField(jsonBody, "output_config.effort"); e.Exists() {
+ if clamped := clampAlibabaMountEffortForModel(model, e.String()); clamped != e.String() {
+ jsonBody, err = providerUtils.SetJSONField(jsonBody, "output_config.effort", clamped)
+ if err != nil {
+ return nil, fmt.Errorf("clamp raw output_config.effort on the model studio mount: %w", err)
+ }
+ }
+ }
+ }
+
// thinking.type — model-gated. Mirrors the typed path in
// stripUnsupportedAnthropicFields; see there for why the legacy
// {"type":"enabled","budget_tokens":N} shape is rewritten to "adaptive"
@@ -682,6 +733,42 @@ func StripUnsupportedFieldsFromRawBody(jsonBody []byte, provider schemas.ModelPr
}
}
+ // Forced-thinking GLM models on z.ai's mount — mirrors the typed path in
+ // stripUnsupportedAnthropicFields: thinking cannot be disabled (GLM-5.3+
+ // 400s), so disabled is rewritten to enabled with the minimum budget, and
+ // on GLM-5.3+ (which requires the field outright — absent = disabled =
+ // 1210 error) a missing thinking field is synthesized, budget derived from
+ // output_config.effort when present.
+ if provider == schemas.Zhipu {
+ thinkingType := providerUtils.GetJSONField(jsonBody, "thinking.type").String()
+ switch {
+ case thinkingType == "disabled" && ZhipuForcedThinkingModel(model):
+ jsonBody, err = providerUtils.SetJSONField(jsonBody, "thinking.type", "enabled")
+ if err != nil {
+ return nil, fmt.Errorf("rewrite raw thinking.type on forced-thinking GLM: %w", err)
+ }
+ jsonBody, err = providerUtils.SetJSONField(jsonBody, "thinking.budget_tokens", MinimumReasoningMaxTokens)
+ if err != nil {
+ return nil, fmt.Errorf("set raw thinking.budget_tokens on forced-thinking GLM: %w", err)
+ }
+ case thinkingType == "" && ZhipuRequiresThinkingModel(model):
+ var effort *string
+ if e := providerUtils.GetJSONField(jsonBody, "output_config.effort"); e.Exists() {
+ v := e.String()
+ effort = &v
+ }
+ maxTokens := int(providerUtils.GetJSONField(jsonBody, "max_tokens").Int())
+ thinking := map[string]interface{}{
+ "type": "enabled",
+ "budget_tokens": zhipuThinkingBudget(effort, maxTokens),
+ }
+ jsonBody, err = providerUtils.SetJSONField(jsonBody, "thinking", thinking)
+ if err != nil {
+ return nil, fmt.Errorf("synthesize raw thinking on GLM-5.3+: %w", err)
+ }
+ }
+ }
+
// top-level cache_control.scope
if !features.PromptCachingScope && providerUtils.JSONFieldExists(jsonBody, "cache_control.scope") {
jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "cache_control.scope")
@@ -1062,6 +1149,219 @@ func DefaultSupportsNativeEffort(model string) bool {
return false
}
+// glm5Minor returns the GLM-5.x minor revision of a (lowercased) model name,
+// or -1 when the model is not a glm-5.... shape. Covers Coding Plan
+// aliases (glm-5.2[1m]) via the leading-digit parse. Mirrors the helper of the
+// same name in core/providers/openai/utils.go — keep the two in lockstep.
+func glm5Minor(modelLower string) int {
+ rest, ok := strings.CutPrefix(modelLower, "glm-5.")
+ if !ok {
+ return -1
+ }
+ digits := 0
+ for digits < len(rest) && rest[digits] >= '0' && rest[digits] <= '9' {
+ digits++
+ }
+ if digits == 0 {
+ return -1
+ }
+ minor, err := strconv.Atoi(rest[:digits])
+ if err != nil {
+ return -1
+ }
+ return minor
+}
+
+// bareModelName lowercases model and strips any provider/routing prefix so the
+// vendor model predicates below match the same strings the upstream sees.
+func bareModelName(model string) string {
+ m := strings.ToLower(model)
+ if i := strings.LastIndex(m, "/"); i >= 0 {
+ m = m[i+1:]
+ }
+ return m
+}
+
+// ZhipuForcedThinkingModel reports whether a GLM model cannot disable thinking
+// on z.ai: GLM-5.3+ errors on thinking.type:"disabled" ("GLM-5.3 no longer
+// supports disabling thinking"), and GLM-4.7 / GLM-4.5V are forced-thinking
+// (disabled is not honored).
+//
+// Source: https://docs.z.ai/guides/capabilities/thinking (verified 2026-08-16)
+func ZhipuForcedThinkingModel(model string) bool {
+ m := bareModelName(model)
+ if glm5Minor(m) >= 3 {
+ return true
+ }
+ return strings.HasPrefix(m, "glm-4.7") || strings.HasPrefix(m, "glm-4.5v")
+}
+
+// ZhipuRequiresThinkingModel reports whether a GLM model requires an explicit
+// thinking:{type:"enabled"} field on z.ai's Anthropic mount: the mount treats an
+// absent thinking field as disabled, and GLM-5.3+ answers that with a 1210
+// error ("This model always engages in thinking and cannot be disabled") —
+// verified live 2026-08-16 (no-thinking and effort-only requests both 400).
+// GLM-4.7 tolerates an absent field (Claude Code's default traffic against the
+// mount carries none), so the requirement starts at 5.3.
+func ZhipuRequiresThinkingModel(model string) bool {
+ return glm5Minor(bareModelName(model)) >= 3
+}
+
+// zhipuThinkingBudget picks the budget_tokens value for a thinking field the
+// gateway rewrites or synthesizes on z.ai's mount: the caller's effort derives
+// one when present, otherwise the minimum — the closest legal shape to "think
+// as little as possible" when the caller gave no reasoning signal.
+func zhipuThinkingBudget(effort *string, maxTokens int) int {
+ budget := MinimumReasoningMaxTokens
+ if effort != nil {
+ if derived, err := providerUtils.GetBudgetTokensFromReasoningEffort(*effort, MinimumReasoningMaxTokens, maxTokens); err == nil {
+ budget = derived
+ }
+ }
+ if maxTokens > 1 && budget >= maxTokens {
+ budget = maxTokens - 1
+ }
+ return budget
+}
+
+// clampAlibabaMountEffortForModel maps an effort value onto the officially
+// valid reasoning_effort enum Model Studio's Anthropic mount accepts for a
+// given model family. The mount proxies Messages to chat-completions
+// internally (errors surface with a chatcmpl-* id) and validates
+// output_config.effort against the per-model reasoning_effort ladder —
+// rejecting out-of-enum values outright ("'reasoning_effort' must be one of:
+// ...") instead of mapping them server-side, so the gateway emits only
+// officially valid values itself. Per the vendor API docs (supplied
+// 2026-08-23; authoritative):
+//
+// - qwen3.8-max: valid xhigh/medium/low. max→xhigh; every other value
+// forwards verbatim.
+// - GLM-5.3+ (the revision that narrowed the enum; the vendor's matrix
+// groups kimi-k3 here, but it never reaches this function — see below):
+// valid max/high/low. xhigh→max, medium→high, minimal→low, none→low.
+// - glm-5.2/glm-5.1/glm-5: valid high/max. xhigh→max, low→high,
+// medium→high, minimal→high, none→high — `low` is out of this family's
+// enum, so the two mildest tiers collapse onto the mildest valid one.
+// - deepseek-v4-pro-0813 / deepseek-v4-flash-0731 (dated snapshots): valid
+// max/high/low. xhigh→high, medium→high, minimal→low, none→low.
+// - deepseek-v4-pro / deepseek-v4-flash (non-dated): same as the GLM-5.2
+// family (valid high/max).
+// - Anything else: verbatim. kimi-k3 never reaches here — its effort is
+// stripped upstream by providerSupportsEffortModel.
+func clampAlibabaMountEffortForModel(model, effort string) string {
+ m := bareModelName(model)
+ switch {
+ case strings.HasPrefix(m, "qwen3.8-max"):
+ if effort == "max" {
+ return "xhigh"
+ }
+ return effort
+ case strings.HasPrefix(m, "deepseek-v4-pro-0813"), strings.HasPrefix(m, "deepseek-v4-flash-0731"):
+ switch effort {
+ case "xhigh", "medium":
+ return "high"
+ case "minimal", "none":
+ return "low"
+ }
+ return effort
+ case glm5Minor(m) >= 3:
+ switch effort {
+ case "xhigh":
+ return "max"
+ case "medium":
+ return "high"
+ case "minimal", "none":
+ return "low"
+ }
+ return effort
+ case glm5Minor(m) >= 0,
+ strings.HasPrefix(m, "deepseek-v4-pro"),
+ strings.HasPrefix(m, "deepseek-v4-flash"):
+ switch effort {
+ case "xhigh":
+ return "max"
+ case "low", "medium", "minimal", "none":
+ // `low` is out of this family's enum (high/max only), so the
+ // mildest tiers all collapse onto the mildest valid value.
+ return "high"
+ }
+ return effort
+ }
+ return effort
+}
+
+// providerSupportsEffortModel scopes output_config.effort to the model families
+// each non-Anthropic vendor documents on its Anthropic-compatible mount. It only
+// runs for providers whose ProviderFeatures entry sets OutputConfigEffort.
+//
+// - Zhipu: z.ai documents reasoning_effort for "GLM-5.2 and above"; on the
+// Coding Plan Anthropic mount out-of-scale values are mapped server-side
+// (GLM-5.3: none/minimal/low→low, medium/high→high, xhigh/max→max), so any
+// Bifrost effort value is safe to forward verbatim.
+// - Alibaba: the mount's API page documents no effort field at all (see Q in
+// types.go); live verification (2026-08-23) shows the mount proxies Messages
+// to its OpenAI-dialect backend, which validates output_config.effort against
+// the per-model reasoning_effort ladder documented on the Model Studio model
+// page — it does NOT map out-of-enum values server-side (max 400s). The
+// gateway therefore clamps to each family's valid values
+// (clampAlibabaMountEffortForModel) at every emission site.
+//
+// Cites: Z on the OutputConfigEffort flag (types.go); the alibaba mount is an
+// empirical contract (live-verified 2026-08-23); per-model matrix from the
+// vendor model-page docs supplied 2026-08-23.
+func providerSupportsEffortModel(provider schemas.ModelProvider, model string) bool {
+ m := bareModelName(model)
+ switch provider {
+ case schemas.Zhipu:
+ return glm5Minor(m) >= 2
+ case schemas.Alibaba:
+ return strings.HasPrefix(m, "qwen3.8-max") ||
+ glm5Minor(m) >= 2 ||
+ strings.HasPrefix(m, "deepseek-v4-pro") ||
+ strings.HasPrefix(m, "deepseek-v4-flash")
+ default:
+ return true
+ }
+}
+
+// SupportsProviderEffort is the provider-aware form of SupportsEffortParameter.
+// Anthropic-family providers keep the model-gated Anthropic behavior; providers
+// whose mount documents the field as a vendor extension (OutputConfigEffort
+// flag) accept it for the vendor-documented model families instead.
+func SupportsProviderEffort(provider schemas.ModelProvider, model string) bool {
+ if features, ok := ProviderFeatures[provider]; ok && features.OutputConfigEffort {
+ return providerSupportsEffortModel(provider, model)
+ }
+ caps := schemas.ResolveModelCaps(provider, model)
+ return caps.SupportsNativeEffort(DefaultSupportsNativeEffort(caps.Model()))
+}
+
+// ResolveAnthropicMountProfile maps well-known third-party Anthropic-compatible
+// hosts to the provider whose capability profile applies. Custom providers
+// (base_provider_type: anthropic) can point at any Anthropic-shaped mount, and
+// gating everything on schemas.Anthropic applies Anthropic's own model rules to
+// vendors whose mounts differ: z.ai accepts output_config.effort (and requires
+// thinking on GLM-5.3+), Model Studio accepts it on specific families, Kimi's
+// mount takes neither. Host-matched (not URL-parsed) so scheme-less base URLs
+// and workspace-dedicated hosts ({WorkspaceId}.{region}.maas.aliyuncs.com)
+// resolve the same way.
+//
+// Verified 2026-08-16 against vendor docs and live upstream behavior; see
+// docs/research 02 §8.4, 03 §6.3, 04 §7.4.
+func ResolveAnthropicMountProfile(baseURL string) schemas.ModelProvider {
+ u := strings.ToLower(baseURL)
+ switch {
+ case strings.Contains(u, "api.z.ai"), strings.Contains(u, "open.bigmodel.cn"):
+ return schemas.Zhipu
+ case strings.Contains(u, "aliyuncs.com"):
+ return schemas.Alibaba
+ case strings.Contains(u, "api.moonshot.ai"), strings.Contains(u, "api.moonshot.cn"), strings.Contains(u, "api.kimi.com"):
+ return schemas.Kimi
+ default:
+ return schemas.Anthropic
+ }
+}
+
// appendToSystemContent merges newContent into existing.
// If existing is nil the new content is returned as-is (preserving ContentStr
// vs ContentBlocks wire format). When both sides are non-empty both are
@@ -1403,12 +1703,24 @@ func MapBifrostEffortToAnthropic(effort string) string {
}
// setEffortOnOutputConfig merges the effort value into the request's OutputConfig,
-// preserving any existing Format field (used for structured outputs).
-func setEffortOnOutputConfig(req *AnthropicMessageRequest, effort string) {
+// preserving any existing Format field (used for structured outputs). Provider-aware:
+// the alibaba mount rejects out-of-enum effort values (clampAlibabaMountEffortForModel),
+// so the value is clamped to the model family's officially valid enum before it lands
+// on the wire.
+func setEffortOnOutputConfig(req *AnthropicMessageRequest, provider schemas.ModelProvider, model, effort string) {
if req.OutputConfig == nil {
req.OutputConfig = &AnthropicOutputConfig{}
}
- req.OutputConfig.Effort = &effort
+ req.OutputConfig.Effort = schemas.Ptr(clampAlibabaMountEffortFor(provider, model, effort))
+}
+
+// clampAlibabaMountEffortFor is the provider-gated form of
+// clampAlibabaMountEffortForModel: only the alibaba profile remaps the value.
+func clampAlibabaMountEffortFor(provider schemas.ModelProvider, model, effort string) string {
+ if provider == schemas.Alibaba {
+ return clampAlibabaMountEffortForModel(model, effort)
+ }
+ return effort
}
// AddMissingBetaHeadersToContext analyzes the Anthropic request and adds missing beta headers to the context.
diff --git a/core/providers/kimi/cachedcontents.go b/core/providers/kimi/cachedcontents.go
new file mode 100644
index 00000000000..adffab48f98
--- /dev/null
+++ b/core/providers/kimi/cachedcontents.go
@@ -0,0 +1,34 @@
+package kimi
+
+import (
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ "github.com/maximhq/bifrost/core/schemas"
+)
+
+// CachedContentCreate is unsupported on KimiProvider. Only Gemini and Vertex AI
+// implement the cached-content lifecycle (Google AI Studio + Vertex AI named
+// caches). Kimi handles caching implicitly (automatic context caching steered by
+// prompt_cache_key).
+func (provider *KimiProvider) CachedContentCreate(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCachedContentCreateRequest) (*schemas.BifrostCachedContentCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentCreateRequest, provider.GetProviderKey())
+}
+
+// CachedContentList is unsupported on KimiProvider (see CachedContentCreate).
+func (provider *KimiProvider) CachedContentList(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentListRequest) (*schemas.BifrostCachedContentListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentListRequest, provider.GetProviderKey())
+}
+
+// CachedContentRetrieve is unsupported on KimiProvider (see CachedContentCreate).
+func (provider *KimiProvider) CachedContentRetrieve(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentRetrieveRequest) (*schemas.BifrostCachedContentRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentRetrieveRequest, provider.GetProviderKey())
+}
+
+// CachedContentUpdate is unsupported on KimiProvider (see CachedContentCreate).
+func (provider *KimiProvider) CachedContentUpdate(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentUpdateRequest) (*schemas.BifrostCachedContentUpdateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentUpdateRequest, provider.GetProviderKey())
+}
+
+// CachedContentDelete is unsupported on KimiProvider (see CachedContentCreate).
+func (provider *KimiProvider) CachedContentDelete(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentDeleteRequest) (*schemas.BifrostCachedContentDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentDeleteRequest, provider.GetProviderKey())
+}
diff --git a/core/providers/kimi/kimi.go b/core/providers/kimi/kimi.go
new file mode 100644
index 00000000000..2e33d944d7b
--- /dev/null
+++ b/core/providers/kimi/kimi.go
@@ -0,0 +1,513 @@
+// Package kimi implements the Kimi (Moonshot AI) LLM provider.
+//
+// Kimi exposes two independent platforms with separate key systems (doc: docs/research
+// 03-provider-kimi.md): the pay-as-you-go Kimi Open Platform (api.moonshot.ai / .cn,
+// the default base URL) and the Kimi Code subscription (api.kimi.com/coding/v1).
+// Both platforms serve an OpenAI-compatible mount (default) and an Anthropic-compatible
+// mount selected per key/alias via use_anthropic_endpoints.
+package kimi
+
+import (
+ "context"
+ "maps"
+ "strings"
+ "time"
+
+ "github.com/maximhq/bifrost/core/providers/anthropic"
+ "github.com/maximhq/bifrost/core/providers/openai"
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ schemas "github.com/maximhq/bifrost/core/schemas"
+ "github.com/valyala/fasthttp"
+)
+
+// KimiProvider implements the Provider interface for Kimi's API.
+type KimiProvider struct {
+ logger schemas.Logger // Logger for provider operations
+ client *fasthttp.Client // HTTP client for unary API requests (ReadTimeout bounds overall response)
+ streamingClient *fasthttp.Client // HTTP client for streaming API requests (no ReadTimeout; idle governed by NewIdleTimeoutReader)
+ networkConfig schemas.NetworkConfig // Network configuration including extra headers
+ sendBackRawRequest bool // Whether to include raw request in BifrostResponse
+ sendBackRawResponse bool // Whether to include raw response in BifrostResponse
+}
+
+// NewKimiProvider creates a new Kimi provider instance.
+// It initializes the HTTP client with the provided configuration and sets up response pools.
+// The client is configured with timeouts, concurrency limits, and optional proxy settings.
+func NewKimiProvider(config *schemas.ProviderConfig, logger schemas.Logger) (*KimiProvider, error) {
+ config.CheckAndSetDefaults()
+
+ // Clone the NetworkConfig (including its mutable maps) so the provider never
+ // shares state with the caller's ProviderConfig — later mutations of the
+ // caller's ExtraHeaders/BetaHeaderOverrides must not affect live requests,
+ // and the BaseURL defaulting below must not write back to the caller.
+ networkConfig := config.NetworkConfig
+ networkConfig.ExtraHeaders = maps.Clone(config.NetworkConfig.ExtraHeaders)
+ networkConfig.BetaHeaderOverrides = maps.Clone(config.NetworkConfig.BetaHeaderOverrides)
+
+ requestTimeout := time.Second * time.Duration(networkConfig.DefaultRequestTimeoutInSeconds)
+ client := &fasthttp.Client{
+ ReadTimeout: requestTimeout,
+ WriteTimeout: requestTimeout,
+ MaxConnsPerHost: networkConfig.MaxConnsPerHost,
+ MaxIdleConnDuration: time.Second * time.Duration(networkConfig.KeepAliveTimeoutInSeconds),
+ MaxConnWaitTimeout: requestTimeout,
+ MaxConnDuration: time.Second * time.Duration(schemas.DefaultMaxConnDurationInSeconds),
+ ConnPoolStrategy: fasthttp.FIFO,
+ }
+
+ // Configure proxy and retry policy
+ client = providerUtils.ConfigureProxy(client, config.ProxyConfig, logger)
+ client = providerUtils.ConfigureDialer(client, networkConfig.AllowPrivateNetwork)
+ client = providerUtils.ConfigureTLS(client, networkConfig, logger)
+ streamingClient := providerUtils.BuildStreamingClient(client)
+ // Set default BaseURL if not provided
+ if networkConfig.BaseURL == "" {
+ networkConfig.BaseURL = defaultBaseURL
+ }
+ networkConfig.BaseURL = strings.TrimRight(networkConfig.BaseURL, "/")
+
+ return &KimiProvider{
+ logger: logger,
+ client: client,
+ streamingClient: streamingClient,
+ networkConfig: networkConfig,
+ sendBackRawRequest: config.SendBackRawRequest,
+ sendBackRawResponse: config.SendBackRawResponse,
+ }, nil
+}
+
+// bearerHeaders builds the Bearer auth headers for Kimi's Anthropic-compatible mount
+// (both mounts authenticate with Authorization: Bearer).
+func (provider *KimiProvider) bearerHeaders(key schemas.Key) map[string]string {
+ headers := map[string]string{}
+ if key.Value.GetValue() != "" {
+ headers["Authorization"] = "Bearer " + key.Value.GetValue()
+ }
+ return headers
+}
+
+// anthropicMessagesURL returns the full Anthropic-mount messages URL for this request,
+// derived from the configured OpenAI base URL and honoring per-request path overrides.
+func (provider *KimiProvider) anthropicMessagesURL(ctx *schemas.BifrostContext) string {
+ anthropicBase := deriveAnthropicBaseURL(provider.networkConfig.BaseURL)
+ return anthropicBase + providerUtils.GetPathFromContext(ctx, anthropicMessagesPathFor(anthropicBase))
+}
+
+// GetProviderKey returns the provider identifier for Kimi.
+func (provider *KimiProvider) GetProviderKey() schemas.ModelProvider {
+ return schemas.Kimi
+}
+
+// ListModels performs a list models request to Kimi's OpenAI-compatible API.
+func (provider *KimiProvider) ListModels(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) {
+ return openai.HandleOpenAIListModelsRequest(
+ ctx,
+ provider.client,
+ request,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, modelsPath),
+ keys,
+ provider.networkConfig.ExtraHeaders,
+ provider.GetProviderKey(),
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ )
+}
+
+// TextCompletion is not supported by the Kimi provider.
+func (provider *KimiProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionRequest, provider.GetProviderKey())
+}
+
+// TextCompletionStream is not supported by the Kimi provider.
+func (provider *KimiProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionStreamRequest, provider.GetProviderKey())
+}
+
+// ChatCompletion performs a chat completion request to Kimi's API.
+func (provider *KimiProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ return anthropic.HandleAnthropicChatCompletionRequest(
+ ctx,
+ provider.client,
+ provider.anthropicMessagesURL(ctx),
+ request,
+ anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Kimi,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ },
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ nil,
+ provider.logger,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIChatCompletionRequest(
+ ctx,
+ provider.client,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, chatCompletionsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ )
+}
+
+// ChatCompletionStream performs a streaming chat completion request to Kimi's API.
+// It supports real-time streaming of responses using Server-Sent Events (SSE).
+// Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails.
+func (provider *KimiProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ jsonData, bifrostErr := anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Kimi,
+ IsStreaming: true,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ })
+ if bifrostErr != nil {
+ return nil, bifrostErr
+ }
+
+ return anthropic.HandleAnthropicChatCompletionStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.anthropicMessagesURL(ctx),
+ jsonData,
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ provider.networkConfig.BetaHeaderOverrides,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ schemas.Kimi,
+ postHookRunner,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIChatCompletionStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, chatCompletionsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ schemas.Kimi,
+ postHookRunner,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+}
+
+// Responses performs a Responses API request against Kimi's Anthropic-compatible
+// mount when use_anthropic_endpoints is set, and otherwise falls back to chat
+// completions on the OpenAI-compatible mount (Kimi exposes no /responses endpoint).
+func (provider *KimiProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ return anthropic.HandleAnthropicResponsesRequest(
+ ctx,
+ provider.client,
+ provider.anthropicMessagesURL(ctx),
+ request,
+ anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Kimi,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ },
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ nil,
+ provider.logger,
+ )
+ }
+
+ chatResponse, err := provider.ChatCompletion(ctx, key, request.ToChatRequest())
+ if err != nil {
+ return nil, err
+ }
+
+ return chatResponse.ToBifrostResponsesResponse(), nil
+}
+
+// ResponsesStream performs a streaming Responses API request against Kimi's
+// Anthropic-compatible mount when use_anthropic_endpoints is set, and otherwise
+// falls back to streaming chat completions on the OpenAI-compatible mount.
+func (provider *KimiProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ jsonData, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Kimi,
+ IsStreaming: true,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ })
+ if bifrostErr != nil {
+ return nil, bifrostErr
+ }
+
+ return anthropic.HandleAnthropicResponsesStream(
+ ctx,
+ provider.streamingClient,
+ provider.anthropicMessagesURL(ctx),
+ jsonData,
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ provider.networkConfig.BetaHeaderOverrides,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ postHookRunner,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyIsResponsesToChatCompletionFallback, true)
+ return provider.ChatCompletionStream(
+ ctx,
+ postHookRunner,
+ postHookSpanFinalizer,
+ key,
+ request.ToChatRequest(),
+ )
+}
+
+// Embedding is not supported by the Kimi provider.
+func (provider *KimiProvider) Embedding(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.EmbeddingRequest, provider.GetProviderKey())
+}
+
+// Speech is not supported by the Kimi provider.
+func (provider *KimiProvider) Speech(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostSpeechRequest) (*schemas.BifrostSpeechResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechRequest, provider.GetProviderKey())
+}
+
+// Rerank is not supported by the Kimi provider.
+func (provider *KimiProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostRerankRequest) (*schemas.BifrostRerankResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.RerankRequest, provider.GetProviderKey())
+}
+
+// OCR is not supported by the Kimi provider.
+func (provider *KimiProvider) OCR(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostOCRRequest) (*schemas.BifrostOCRResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.OCRRequest, provider.GetProviderKey())
+}
+
+// SpeechStream is not supported by the Kimi provider.
+func (provider *KimiProvider) SpeechStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostSpeechRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechStreamRequest, provider.GetProviderKey())
+}
+
+// Transcription is not supported by the Kimi provider.
+func (provider *KimiProvider) Transcription(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTranscriptionRequest) (*schemas.BifrostTranscriptionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionRequest, provider.GetProviderKey())
+}
+
+// TranscriptionStream is not supported by the Kimi provider.
+func (provider *KimiProvider) TranscriptionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTranscriptionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionStreamRequest, provider.GetProviderKey())
+}
+
+// ImageGeneration is not supported by the Kimi provider.
+func (provider *KimiProvider) ImageGeneration(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageGenerationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationRequest, provider.GetProviderKey())
+}
+
+// ImageGenerationStream is not supported by the Kimi provider.
+func (provider *KimiProvider) ImageGenerationStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageGenerationRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationStreamRequest, provider.GetProviderKey())
+}
+
+// ImageEdit is not supported by the Kimi provider.
+func (provider *KimiProvider) ImageEdit(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageEditRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditRequest, provider.GetProviderKey())
+}
+
+// ImageEditStream is not supported by the Kimi provider.
+func (provider *KimiProvider) ImageEditStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageEditRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditStreamRequest, provider.GetProviderKey())
+}
+
+// ImageVariation is not supported by the Kimi provider.
+func (provider *KimiProvider) ImageVariation(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageVariationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageVariationRequest, provider.GetProviderKey())
+}
+
+// VideoGeneration is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoGeneration(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoGenerationRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoGenerationRequest, provider.GetProviderKey())
+}
+
+// VideoRetrieve is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoRetrieve(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRetrieveRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRetrieveRequest, provider.GetProviderKey())
+}
+
+// VideoDownload is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoDownload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDownloadRequest) (*schemas.BifrostVideoDownloadResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDownloadRequest, provider.GetProviderKey())
+}
+
+// VideoDelete is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoDelete(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDeleteRequest) (*schemas.BifrostVideoDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDeleteRequest, provider.GetProviderKey())
+}
+
+// VideoList is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoList(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoListRequest) (*schemas.BifrostVideoListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoListRequest, provider.GetProviderKey())
+}
+
+// VideoEdit is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoEdit(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoEditRequest) (*schemas.BifrostVideoEditResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoEditRequest, provider.GetProviderKey())
+}
+
+// VideoRemix is not supported by the Kimi provider.
+func (provider *KimiProvider) VideoRemix(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRemixRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRemixRequest, provider.GetProviderKey())
+}
+
+// FileUpload is not supported by the Kimi provider.
+func (provider *KimiProvider) FileUpload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostFileUploadRequest) (*schemas.BifrostFileUploadResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileUploadRequest, provider.GetProviderKey())
+}
+
+// FileList is not supported by the Kimi provider.
+func (provider *KimiProvider) FileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileListRequest) (*schemas.BifrostFileListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileListRequest, provider.GetProviderKey())
+}
+
+// FileRetrieve is not supported by the Kimi provider.
+func (provider *KimiProvider) FileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileRetrieveRequest) (*schemas.BifrostFileRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileRetrieveRequest, provider.GetProviderKey())
+}
+
+// FileDelete is not supported by the Kimi provider.
+func (provider *KimiProvider) FileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileDeleteRequest) (*schemas.BifrostFileDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileDeleteRequest, provider.GetProviderKey())
+}
+
+// FileContent is not supported by the Kimi provider.
+func (provider *KimiProvider) FileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileContentRequest) (*schemas.BifrostFileContentResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileContentRequest, provider.GetProviderKey())
+}
+
+// BatchCreate is not supported by the Kimi provider.
+func (provider *KimiProvider) BatchCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostBatchCreateRequest) (*schemas.BifrostBatchCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCreateRequest, provider.GetProviderKey())
+}
+
+// BatchList is not supported by the Kimi provider.
+func (provider *KimiProvider) BatchList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchListRequest) (*schemas.BifrostBatchListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchListRequest, provider.GetProviderKey())
+}
+
+// BatchRetrieve is not supported by the Kimi provider.
+func (provider *KimiProvider) BatchRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchRetrieveRequest) (*schemas.BifrostBatchRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchRetrieveRequest, provider.GetProviderKey())
+}
+
+// BatchCancel is not supported by the Kimi provider.
+func (provider *KimiProvider) BatchCancel(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchCancelRequest) (*schemas.BifrostBatchCancelResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCancelRequest, provider.GetProviderKey())
+}
+
+// BatchDelete is not supported by the Kimi provider.
+func (provider *KimiProvider) BatchDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchDeleteRequest) (*schemas.BifrostBatchDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchDeleteRequest, provider.GetProviderKey())
+}
+
+// BatchResults is not supported by the Kimi provider.
+func (provider *KimiProvider) BatchResults(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchResultsRequest) (*schemas.BifrostBatchResultsResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchResultsRequest, provider.GetProviderKey())
+}
+
+// CountTokens is not supported by the Kimi provider. Kimi's token-counting endpoint
+// (/v1/tokenizers/estimate-token-count) uses a Kimi-specific contract that is not
+// covered by the Anthropic-style count-tokens handlers.
+func (provider *KimiProvider) CountTokens(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostCountTokensResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CountTokensRequest, provider.GetProviderKey())
+}
+
+// Compaction is not supported by the Kimi provider.
+func (provider *KimiProvider) Compaction(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCompactionRequest) (*schemas.BifrostCompactionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CompactionRequest, provider.GetProviderKey())
+}
+
+// ContainerCreate is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerCreateRequest) (*schemas.BifrostContainerCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerCreateRequest, provider.GetProviderKey())
+}
+
+// ContainerList is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerListRequest) (*schemas.BifrostContainerListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerListRequest, provider.GetProviderKey())
+}
+
+// ContainerRetrieve is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerRetrieveRequest) (*schemas.BifrostContainerRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerRetrieveRequest, provider.GetProviderKey())
+}
+
+// ContainerDelete is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerDeleteRequest) (*schemas.BifrostContainerDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerDeleteRequest, provider.GetProviderKey())
+}
+
+// ContainerFileCreate is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerFileCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerFileCreateRequest) (*schemas.BifrostContainerFileCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileCreateRequest, provider.GetProviderKey())
+}
+
+// ContainerFileList is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerFileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileListRequest) (*schemas.BifrostContainerFileListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileListRequest, provider.GetProviderKey())
+}
+
+// ContainerFileRetrieve is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerFileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileRetrieveRequest) (*schemas.BifrostContainerFileRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileRetrieveRequest, provider.GetProviderKey())
+}
+
+// ContainerFileContent is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerFileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileContentRequest) (*schemas.BifrostContainerFileContentResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileContentRequest, provider.GetProviderKey())
+}
+
+// ContainerFileDelete is not supported by the Kimi provider.
+func (provider *KimiProvider) ContainerFileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileDeleteRequest) (*schemas.BifrostContainerFileDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileDeleteRequest, provider.GetProviderKey())
+}
+
+// Passthrough is not supported by the Kimi provider.
+func (provider *KimiProvider) Passthrough(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (*schemas.BifrostPassthroughResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughRequest, provider.GetProviderKey())
+}
+
+// PassthroughStream is not supported by the Kimi provider.
+func (provider *KimiProvider) PassthroughStream(_ *schemas.BifrostContext, _ schemas.PostHookRunner, _ func(context.Context), _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughStreamRequest, provider.GetProviderKey())
+}
diff --git a/core/providers/kimi/kimi_test.go b/core/providers/kimi/kimi_test.go
new file mode 100644
index 00000000000..4dfa8c6e12d
--- /dev/null
+++ b/core/providers/kimi/kimi_test.go
@@ -0,0 +1,63 @@
+package kimi_test
+
+import (
+ "os"
+ "strings"
+ "testing"
+
+ "github.com/maximhq/bifrost/core/internal/llmtests"
+
+ "github.com/maximhq/bifrost/core/schemas"
+)
+
+func TestKimi(t *testing.T) {
+ t.Parallel()
+ if strings.TrimSpace(os.Getenv("KIMI_API_KEY")) == "" {
+ t.Skip("Skipping Kimi tests because KIMI_API_KEY is not set")
+ }
+
+ client, ctx, cancel, err := llmtests.SetupTest()
+ if err != nil {
+ t.Fatalf("Error initializing test setup: %v", err)
+ }
+ defer cancel()
+ defer client.Shutdown()
+
+ testConfig := llmtests.ComprehensiveTestConfig{
+ Provider: schemas.Kimi,
+ ChatModel: "kimi-k2.6",
+ Fallbacks: []schemas.Fallback{
+ {Provider: schemas.Kimi, Model: "kimi-k2.6"},
+ {Provider: schemas.Kimi, Model: "kimi-k3"},
+ },
+ TextModel: "kimi-k2.6",
+ EmbeddingModel: "", // Kimi doesn't support embeddings
+ ReasoningModel: "kimi-k3",
+ Scenarios: llmtests.TestScenarios{
+ SimpleChat: true,
+ CompletionStream: true,
+ MultiTurnConversation: true,
+ ToolCalls: true,
+ ToolCallsStreaming: true,
+ End2EndToolCalling: true,
+ AutomaticFunctionCall: true,
+ // Kimi vision accepts base64/ms:// refs only (no public image URLs);
+ // exercise vision during dogfooding instead.
+ ImageURL: false,
+ ImageBase64: false,
+ MultipleImages: false,
+ CompleteEnd2End: true,
+ Embedding: false,
+ ListModels: true,
+ // The Reasoning scenario runs via the Responses API, which Kimi does
+ // not expose — reasoning_effort routing is covered by unit tests and
+ // dogfood smoke tests instead.
+ Reasoning: false,
+ PassThroughExtraParams: true,
+ },
+ }
+
+ t.Run("KimiTests", func(t *testing.T) {
+ llmtests.RunAllComprehensiveTests(t, client, ctx, testConfig)
+ })
+}
diff --git a/core/providers/kimi/utils.go b/core/providers/kimi/utils.go
new file mode 100644
index 00000000000..6ddd8f7d766
--- /dev/null
+++ b/core/providers/kimi/utils.go
@@ -0,0 +1,92 @@
+package kimi
+
+import (
+ "net/url"
+ "strings"
+)
+
+const (
+ // defaultBaseURL is the Kimi Open Platform (international) OpenAI-compatible base URL.
+ // Kimi Code (subscription) users override it with https://api.kimi.com/coding/v1;
+ // CN Open Platform users with https://api.moonshot.cn/v1.
+ defaultBaseURL = "https://api.moonshot.ai/v1"
+
+ // chatCompletionsPath is the chat completions path relative to the OpenAI-compatible base URL.
+ chatCompletionsPath = "/chat/completions"
+
+ // modelsPath is the list-models path relative to the OpenAI-compatible base URL.
+ modelsPath = "/models"
+
+ // openPlatformAnthropicMount is the Anthropic mount suffix on Open Platform hosts
+ // (api.moonshot.ai / api.moonshot.cn): /v1 is replaced by /anthropic.
+ openPlatformAnthropicMount = "/anthropic"
+
+ // anthropicMessagesPath is the messages path under a derived Anthropic mount base.
+ anthropicMessagesPath = "/v1/messages"
+
+ // kimiCodingBaseSuffix marks the Kimi Code OpenAI-compatible base URL. On this
+ // mount the Anthropic endpoint shares the same base (messages at /coding/v1/messages).
+ kimiCodingBaseSuffix = "/coding/v1"
+
+ // kimiCodingAnthropicMessagesPath is the messages path on the Kimi Code Anthropic mount.
+ kimiCodingAnthropicMessagesPath = "/messages"
+)
+
+// kimiKnownHosts are the upstream hosts whose URL shapes the suffix rewrites
+// below rely on. Custom or proxied hosts never get their paths rewritten.
+var kimiKnownHosts = map[string]bool{
+ "api.kimi.com": true, // Kimi Code subscription
+ "api.moonshot.ai": true, // Open Platform (international)
+ "api.moonshot.cn": true, // Open Platform (China)
+}
+
+// isKnownKimiHost reports whether the base URL sits on one of Kimi's own hosts.
+// Unparseable or scheme-less inputs are treated as custom hosts.
+func isKnownKimiHost(base string) bool {
+ parsed, err := url.Parse(base)
+ if err != nil || parsed.Scheme == "" || parsed.Host == "" {
+ return false
+ }
+ return kimiKnownHosts[parsed.Host]
+}
+
+// deriveAnthropicBaseURL derives the Anthropic-compatible mount base URL from the
+// configured OpenAI-compatible base URL.
+//
+// - Open Platform (api.moonshot.ai / api.moonshot.cn): a trailing /v1 is replaced
+// with /anthropic, so messages live at /anthropic/v1/messages.
+// - Kimi Code (api.kimi.com/coding/v1): the Anthropic mount shares the OpenAI base,
+// with messages at /coding/v1/messages.
+// - Any other host — including custom bases that happen to end in /v1 or
+// /coding/v1 — keeps its configured path and gets /anthropic appended, so a
+// proxy's URL shape is never silently rewritten; users on exotic hosts can
+// always create a second provider instance with an explicit base URL.
+//
+// Idempotent: a base that already ends with the mount suffix is returned unchanged —
+// use_anthropic_endpoints with a base_url set to the mount itself must not append
+// /anthropic a second time.
+func deriveAnthropicBaseURL(openAIBaseURL string) string {
+ base := strings.TrimRight(openAIBaseURL, "/")
+ if strings.HasSuffix(base, openPlatformAnthropicMount) {
+ return base
+ }
+ if !isKnownKimiHost(base) {
+ return base + openPlatformAnthropicMount
+ }
+ if strings.HasSuffix(base, kimiCodingBaseSuffix) {
+ return base
+ }
+ if strings.HasSuffix(base, "/v1") {
+ return strings.TrimSuffix(base, "/v1") + openPlatformAnthropicMount
+ }
+ return base + openPlatformAnthropicMount
+}
+
+// anthropicMessagesPathFor returns the messages path to append to the derived
+// Anthropic mount base URL.
+func anthropicMessagesPathFor(anthropicBaseURL string) string {
+ if strings.HasSuffix(anthropicBaseURL, kimiCodingBaseSuffix) {
+ return kimiCodingAnthropicMessagesPath
+ }
+ return anthropicMessagesPath
+}
diff --git a/core/providers/kimi/utils_test.go b/core/providers/kimi/utils_test.go
new file mode 100644
index 00000000000..0b322f3c0f3
--- /dev/null
+++ b/core/providers/kimi/utils_test.go
@@ -0,0 +1,94 @@
+package kimi
+
+import "testing"
+
+func TestDeriveAnthropicBaseURL(t *testing.T) {
+ tests := []struct {
+ name string
+ openAIBase string
+ wantBase string
+ wantMessages string // full messages URL
+ }{
+ {
+ name: "Open Platform international",
+ openAIBase: "https://api.moonshot.ai/v1",
+ wantBase: "https://api.moonshot.ai/anthropic",
+ wantMessages: "https://api.moonshot.ai/anthropic/v1/messages",
+ },
+ {
+ name: "Open Platform China",
+ openAIBase: "https://api.moonshot.cn/v1",
+ wantBase: "https://api.moonshot.cn/anthropic",
+ wantMessages: "https://api.moonshot.cn/anthropic/v1/messages",
+ },
+ {
+ name: "Kimi Code subscription mount shares the OpenAI base",
+ openAIBase: "https://api.kimi.com/coding/v1",
+ wantBase: "https://api.kimi.com/coding/v1",
+ wantMessages: "https://api.kimi.com/coding/v1/messages",
+ },
+ {
+ name: "trailing slash is trimmed",
+ openAIBase: "https://api.moonshot.ai/v1/",
+ wantBase: "https://api.moonshot.ai/anthropic",
+ wantMessages: "https://api.moonshot.ai/anthropic/v1/messages",
+ },
+ {
+ name: "custom base without /v1 falls back to appending /anthropic",
+ openAIBase: "https://proxy.example.com/kimi",
+ wantBase: "https://proxy.example.com/kimi/anthropic",
+ wantMessages: "https://proxy.example.com/kimi/anthropic/v1/messages",
+ },
+ {
+ // Suffix semantics only hold on Kimi's own hosts: a custom host
+ // ending in /v1 must NOT have it rewritten away.
+ name: "custom base ending in /v1 keeps its path",
+ openAIBase: "https://proxy.example.com/v1",
+ wantBase: "https://proxy.example.com/v1/anthropic",
+ wantMessages: "https://proxy.example.com/v1/anthropic/v1/messages",
+ },
+ {
+ // Likewise for the Kimi Code suffix shape on a foreign host: the
+ // base is not assumed to serve the Anthropic mount at its root.
+ name: "custom base ending in /coding/v1 keeps its path",
+ openAIBase: "https://proxy.example.com/coding/v1",
+ wantBase: "https://proxy.example.com/coding/v1/anthropic",
+ wantMessages: "https://proxy.example.com/coding/v1/anthropic/v1/messages",
+ },
+ {
+ // Scheme-less values must take the custom-host fallback even when
+ // the remainder parses to a known Kimi host.
+ name: "scheme-less base takes the custom-host fallback",
+ openAIBase: "//api.kimi.com/v1",
+ wantBase: "//api.kimi.com/v1/anthropic",
+ wantMessages: "//api.kimi.com/v1/anthropic/v1/messages",
+ },
+ {
+ // use_anthropic_endpoints with a base_url already set to the mount
+ // itself must not append /anthropic a second time.
+ name: "Open Platform Anthropic mount as base is idempotent",
+ openAIBase: "https://api.moonshot.ai/anthropic",
+ wantBase: "https://api.moonshot.ai/anthropic",
+ wantMessages: "https://api.moonshot.ai/anthropic/v1/messages",
+ },
+ {
+ name: "custom base already ending in /anthropic keeps its path",
+ openAIBase: "https://proxy.example.com/anthropic",
+ wantBase: "https://proxy.example.com/anthropic",
+ wantMessages: "https://proxy.example.com/anthropic/v1/messages",
+ },
+ }
+
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ base := deriveAnthropicBaseURL(tt.openAIBase)
+ if base != tt.wantBase {
+ t.Errorf("deriveAnthropicBaseURL(%q) = %q, want %q", tt.openAIBase, base, tt.wantBase)
+ }
+ full := base + anthropicMessagesPathFor(base)
+ if full != tt.wantMessages {
+ t.Errorf("messages URL for %q = %q, want %q", tt.openAIBase, full, tt.wantMessages)
+ }
+ })
+ }
+}
diff --git a/core/providers/openai/chat.go b/core/providers/openai/chat.go
index 3cbe24f738d..db120fe8efa 100644
--- a/core/providers/openai/chat.go
+++ b/core/providers/openai/chat.go
@@ -89,6 +89,30 @@ func ToOpenAIChatRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.Bifros
// tool-calling conversation — see issue #5887.
openaiReq.stripReasoningDetailsExceptToolCalls()
return openaiReq
+ case schemas.Kimi:
+ // Kimi natively supports prediction and prompt_cache_key — the generic filter
+ // strips both, so preserve them.
+ prediction := openaiReq.ChatParameters.Prediction
+ promptCacheKey := openaiReq.ChatParameters.PromptCacheKey
+ openaiReq.filterOpenAISpecificParameters(caps)
+ openaiReq.ChatParameters.Prediction = prediction
+ openaiReq.ChatParameters.PromptCacheKey = promptCacheKey
+ // Reasoning content must be replayed verbatim on K-series assistant turns
+ // (Preserved Thinking), so do NOT strip reasoning here.
+ openaiReq.applyKimiReasoning(capModel)
+ return openaiReq
+ case schemas.Zhipu:
+ openaiReq.filterOpenAISpecificParameters(caps)
+ // GLM-5.2+ and GLM-4.7 (forced thinking) rely on reasoning_content replay for
+ // interleaved/preserved thinking — do NOT strip reasoning here.
+ openaiReq.applyZhipuReasoning(capModel)
+ return openaiReq
+ case schemas.Alibaba:
+ openaiReq.filterOpenAISpecificParameters(caps)
+ // qwen3.8-max (+ hosted deepseek-v4/glm) replay reasoning_content on multi-turn
+ // thinking flows — do NOT strip reasoning here.
+ openaiReq.applyAlibabaReasoning(capModel)
+ return openaiReq
case schemas.XAI:
openaiReq.filterOpenAISpecificParameters(caps)
openaiReq.applyXAICompatibility(caps)
@@ -299,3 +323,113 @@ func (req *OpenAIChatRequest) applyXAICompatibility(caps schemas.ModelCaps) {
req.ChatParameters.Reasoning.Effort = nil
}
}
+
+// applyKimiReasoning routes ingress reasoning.effort to Kimi's top-level
+// reasoning_effort field. Only K3-class models (kimi-k3 and the Kimi Code aliases
+// k3 / k3-256k) accept the field (low/high/max); every other Kimi model rejects
+// it outright (K2.x thinking is controlled via the `thinking` extra param instead).
+// Runs after filterOpenAISpecificParameters, so effort values are already
+// normalized (minimal→low; max preserved for kimi-k3 via supportsMaxReasoningEffort;
+// xhigh→high — a safe downgrade for K3, which 400s on xhigh).
+func (req *OpenAIChatRequest) applyKimiReasoning(capModel string) {
+ if req.ChatParameters.Reasoning == nil {
+ return
+ }
+ if !isKimiK3Model(capModel) {
+ req.ChatParameters.Reasoning = nil
+ return
+ }
+ if req.ChatParameters.Reasoning.Effort == nil {
+ return
+ }
+ switch *req.ChatParameters.Reasoning.Effort {
+ case "none":
+ // K3 always thinks and 400s on "none" — drop the field so the model
+ // default applies instead of surfacing a vendor error.
+ req.ChatParameters.Reasoning = nil
+ case "medium":
+ req.ChatParameters.Reasoning.Effort = schemas.Ptr("high")
+ }
+}
+
+// isKimiK3Model reports whether the model accepts top-level reasoning_effort
+// (kimi-k3 plus the Kimi Code stable aliases k3 / k3-256k).
+func isKimiK3Model(model string) bool {
+ _, parsedModel := schemas.ParseModelString(model, schemas.Kimi)
+ if parsedModel != "" {
+ model = parsedModel
+ }
+ modelLower := strings.ToLower(model)
+ return strings.HasPrefix(modelLower, "kimi-k3") ||
+ modelLower == "k3" ||
+ strings.HasPrefix(modelLower, "k3-")
+}
+
+// applyZhipuReasoning keeps reasoning_effort only for GLM-5.2+ — the Zhipu
+// series that honors it. GLM-5.2 accepts the full legacy enum and maps the
+// extra tiers itself (low/medium→high, xhigh→max). GLM-5.3 narrowed the enum
+// to exactly max/high/low and rejects every other value on the General API,
+// so for GLM-5.3+ the wider tiers are mapped onto the nearest supported one,
+// mirroring the Coding Plan's own coercion table (xhigh→max, medium→high,
+// minimal/none→low). GLM-4.x ignore or 400 the field; their thinking is
+// controlled via the `thinking` extra param instead.
+func (req *OpenAIChatRequest) applyZhipuReasoning(capModel string) {
+ if req.ChatParameters.Reasoning == nil {
+ return
+ }
+ _, parsedModel := schemas.ParseModelString(capModel, schemas.Zhipu)
+ if parsedModel != "" {
+ capModel = parsedModel
+ }
+ modelLower := strings.ToLower(capModel)
+ if !isGLM52OrLater(modelLower) {
+ req.ChatParameters.Reasoning = nil
+ return
+ }
+ effort := req.ChatParameters.Reasoning.Effort
+ if effort == nil || !isGLM53OrLater(modelLower) {
+ return
+ }
+ switch *effort {
+ case "max", "high", "low":
+ // Already in GLM-5.3's accepted set.
+ case "xhigh":
+ req.ChatParameters.Reasoning.Effort = schemas.Ptr("max")
+ case "medium":
+ req.ChatParameters.Reasoning.Effort = schemas.Ptr("high")
+ default:
+ // "minimal", "none", and anything else legacy — the mildest tier.
+ req.ChatParameters.Reasoning.Effort = schemas.Ptr("low")
+ }
+}
+
+// applyAlibabaReasoning keeps reasoning_effort only for the Model Studio models
+// that honor it: qwen3.8-max and hosted DeepSeek-V4 / GLM-5 series. On the
+// OpenAI-compatible mount qwen3.8-max's native enum tops out at xhigh
+// (none/minimal/low/medium/high/xhigh — the vendor 400'd "max" until ~2026-08-22),
+// so the shared normalizer (which runs before this pass) forwards xhigh verbatim,
+// clamps max down to xhigh, and maps minimal→low; it does not rely on any
+// vendor-side auto-mapping. Every other model rejects the field — thinking there
+// is controlled via the enable_thinking / thinking_budget extra params instead.
+func (req *OpenAIChatRequest) applyAlibabaReasoning(capModel string) {
+ if req.ChatParameters.Reasoning == nil {
+ return
+ }
+ if !isAlibabaReasoningEffortModel(capModel) {
+ req.ChatParameters.Reasoning = nil
+ }
+}
+
+// isAlibabaReasoningEffortModel reports whether the model accepts reasoning_effort
+// on the Model Studio OpenAI-compatible mount.
+func isAlibabaReasoningEffortModel(model string) bool {
+ _, parsedModel := schemas.ParseModelString(model, schemas.Alibaba)
+ if parsedModel != "" {
+ model = parsedModel
+ }
+ modelLower := strings.ToLower(model)
+ return strings.HasPrefix(modelLower, "qwen3.8-max") ||
+ strings.HasPrefix(modelLower, "deepseek-v4") ||
+ // Hosted GLM-5 series (glm-5, glm-5.1, glm-5.2, and later revisions).
+ strings.HasPrefix(modelLower, "glm-5")
+}
diff --git a/core/providers/openai/chat_test.go b/core/providers/openai/chat_test.go
index 199304c861e..33b074035e7 100644
--- a/core/providers/openai/chat_test.go
+++ b/core/providers/openai/chat_test.go
@@ -329,6 +329,57 @@ func TestToOpenAIChatRequest_NormalizesReasoningEffort(t *testing.T) {
effort: "max",
expected: "max",
},
+ {
+ name: "preserves max for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "max",
+ expected: "max",
+ },
+ {
+ name: "preserves max for provider-prefixed glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "zai/glm-5.3",
+ effort: "max",
+ expected: "max",
+ },
+ {
+ // GLM-5.3 only accepts max/high/low; wider tiers are mapped.
+ name: "maps medium to high for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "medium",
+ expected: "high",
+ },
+ {
+ name: "maps none to low for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "none",
+ expected: "low",
+ },
+ {
+ name: "maps xhigh to max for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "xhigh",
+ expected: "max",
+ },
+ {
+ name: "maps minimal to low for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "minimal",
+ expected: "low",
+ },
+ {
+ // GLM-5.2 keeps the full legacy enum (the vendor maps it itself).
+ name: "keeps medium for glm-5.2",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.2",
+ effort: "medium",
+ expected: "medium",
+ },
}
for _, tt := range tests {
@@ -597,6 +648,30 @@ func TestOpenAIChatRequest_FilterOpenAISpecificParameters_NormalizesReasoningEff
effort: "max",
expected: "max",
},
+ {
+ name: "preserves max for glm-5.3",
+ model: "glm-5.3",
+ effort: "max",
+ expected: "max",
+ },
+ {
+ name: "preserves max for provider-prefixed glm-5.3",
+ model: "zai/glm-5.3",
+ effort: "max",
+ expected: "max",
+ },
+ {
+ name: "maps medium to high for glm-5.3",
+ model: "glm-5.3",
+ effort: "medium",
+ expected: "high",
+ },
+ {
+ name: "maps none to low for glm-5.3",
+ model: "zai/glm-5.3",
+ effort: "none",
+ expected: "low",
+ },
}
for _, tt := range tests {
@@ -2039,3 +2114,217 @@ func TestOpenAICompatFiltersReadDatasheet(t *testing.T) {
require.Nil(t, req.FrequencyPenalty, "fields the row omits keep the grok name-based default")
})
}
+
+// Kimi routes reasoning.effort to top-level reasoning_effort only for K3-class
+// models (kimi-k3 and the Kimi Code aliases k3 / k3-256k); every other Kimi model
+// rejects the field. K3 accepts low/high/max — "none" is dropped (K3 always
+// thinks) and "medium" maps to "high". Regression: effort "max" on kimi-k3 must
+// arrive as "max" (not silently downgraded to "high").
+func TestToOpenAIChatRequest_KimiReasoningRouting(t *testing.T) {
+ ctx, cancel := schemas.NewBifrostContextWithCancel(nil)
+ defer cancel()
+
+ tests := []struct {
+ name string
+ model string
+ effort string
+ wantEffort *string // nil = reasoning must be stripped entirely
+ }{
+ {name: "kimi-k3 keeps max", model: "kimi-k3", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "kimi-k3 keeps low", model: "kimi-k3", effort: "low", wantEffort: schemas.Ptr("low")},
+ {name: "kimi-k3 maps medium to high", model: "kimi-k3", effort: "medium", wantEffort: schemas.Ptr("high")},
+ {name: "kimi-k3 maps minimal to low", model: "kimi-k3", effort: "minimal", wantEffort: schemas.Ptr("low")},
+ {name: "kimi-k3 drops none", model: "kimi-k3", effort: "none", wantEffort: nil},
+ {name: "Kimi Code alias k3 keeps max", model: "k3", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "Kimi Code alias k3-256k keeps high", model: "k3-256k", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "provider-prefixed model still routes", model: "kimi/kimi-k3", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "kimi-k2.7-code strips effort", model: "kimi-k2.7-code", effort: "high", wantEffort: nil},
+ {name: "kimi-k2.6 strips effort", model: "kimi-k2.6", effort: "max", wantEffort: nil},
+ {name: "kimi-k2.5 strips effort", model: "kimi-k2.5", effort: "low", wantEffort: nil},
+ {name: "moonshot-v1 strips effort", model: "moonshot-v1-8k", effort: "high", wantEffort: nil},
+ }
+
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ out := ToOpenAIChatRequest(ctx, &schemas.BifrostChatRequest{
+ Provider: schemas.Kimi,
+ Model: tt.model,
+ Input: []schemas.ChatMessage{{
+ Role: schemas.ChatMessageRoleUser,
+ Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hello")},
+ }},
+ Params: &schemas.ChatParameters{
+ Reasoning: &schemas.ChatReasoning{Effort: schemas.Ptr(tt.effort)},
+ },
+ })
+ require.NotNil(t, out)
+
+ if tt.wantEffort == nil {
+ if out.ChatParameters.Reasoning != nil && out.ChatParameters.Reasoning.Effort != nil {
+ t.Fatalf("expected reasoning to be stripped, got effort %q", *out.ChatParameters.Reasoning.Effort)
+ }
+ return
+ }
+ require.NotNil(t, out.ChatParameters.Reasoning, "expected reasoning to be kept")
+ require.NotNil(t, out.ChatParameters.Reasoning.Effort, "expected effort to be kept")
+ require.Equal(t, *tt.wantEffort, *out.ChatParameters.Reasoning.Effort)
+ })
+ }
+}
+
+// Kimi natively supports prediction and prompt_cache_key; the generic
+// OpenAI-specific filter would strip both, so the Kimi case must preserve them.
+func TestToOpenAIChatRequest_KimiPreservesPredictionAndCacheKey(t *testing.T) {
+ ctx, cancel := schemas.NewBifrostContextWithCancel(nil)
+ defer cancel()
+
+ prediction := &schemas.ChatPrediction{Type: "content", Content: "the capital of France is"}
+ cacheKey := "session-42"
+ safety := "user-hash-7"
+
+ out := ToOpenAIChatRequest(ctx, &schemas.BifrostChatRequest{
+ Provider: schemas.Kimi,
+ Model: "kimi-k2.6",
+ Input: []schemas.ChatMessage{{
+ Role: schemas.ChatMessageRoleUser,
+ Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hello")},
+ }},
+ Params: &schemas.ChatParameters{
+ Prediction: prediction,
+ PromptCacheKey: &cacheKey,
+ SafetyIdentifier: &safety,
+ },
+ })
+ require.NotNil(t, out)
+ require.Same(t, prediction, out.ChatParameters.Prediction, "prediction must survive the Kimi case")
+ require.NotNil(t, out.ChatParameters.PromptCacheKey)
+ require.Equal(t, cacheKey, *out.ChatParameters.PromptCacheKey)
+ require.NotNil(t, out.ChatParameters.SafetyIdentifier, "safety_identifier is first-class and must pass through")
+ require.Equal(t, safety, *out.ChatParameters.SafetyIdentifier)
+}
+
+// Zhipu keeps reasoning_effort only for GLM-5.2+ (full enum, incl. the Coding
+// Plan 1M-context alias glm-5.2[1m]); every other GLM model ignores or 400s
+// the field. The 5.2+ family is version-floored, so future revisions
+// (glm-5.3, glm-5.5, ...) are covered on day one.
+func TestToOpenAIChatRequest_ZhipuReasoningRouting(t *testing.T) {
+ ctx, cancel := schemas.NewBifrostContextWithCancel(nil)
+ defer cancel()
+
+ tests := []struct {
+ name string
+ model string
+ effort string
+ wantEffort *string
+ }{
+ {name: "glm-5.2 keeps max", model: "glm-5.2", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "glm-5.2 keeps none", model: "glm-5.2", effort: "none", wantEffort: schemas.Ptr("none")},
+ {name: "glm-5.2[1m] keeps high", model: "glm-5.2[1m]", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "glm-5.2 keeps medium (vendor maps it)", model: "glm-5.2", effort: "medium", wantEffort: schemas.Ptr("medium")},
+ {name: "glm-5.3 keeps max", model: "glm-5.3", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "glm-5.3 keeps high", model: "glm-5.3", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "glm-5.3 keeps low", model: "glm-5.3", effort: "low", wantEffort: schemas.Ptr("low")},
+ {name: "glm-5.3 maps medium to high", model: "glm-5.3", effort: "medium", wantEffort: schemas.Ptr("high")},
+ {name: "glm-5.3 maps none to low", model: "glm-5.3", effort: "none", wantEffort: schemas.Ptr("low")},
+ {name: "glm-5.3 maps minimal to low", model: "glm-5.3", effort: "minimal", wantEffort: schemas.Ptr("low")},
+ {name: "glm-5.3 maps xhigh to max", model: "glm-5.3", effort: "xhigh", wantEffort: schemas.Ptr("max")},
+ {name: "glm-5.5 maps medium to high", model: "glm-5.5", effort: "medium", wantEffort: schemas.Ptr("high")},
+ {name: "glm-5.2-air keeps medium", model: "glm-5.2-air", effort: "medium", wantEffort: schemas.Ptr("medium")},
+ {name: "glm-5.1 strips effort", model: "glm-5.1", effort: "high", wantEffort: nil},
+ {name: "glm-4.7 strips effort", model: "glm-4.7", effort: "high", wantEffort: nil},
+ {name: "glm-4.5-flash strips effort", model: "glm-4.5-flash", effort: "low", wantEffort: nil},
+ {name: "glm-4.6v strips effort", model: "glm-4.6v", effort: "medium", wantEffort: nil},
+ }
+
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ out := ToOpenAIChatRequest(ctx, &schemas.BifrostChatRequest{
+ Provider: schemas.Zhipu,
+ Model: tt.model,
+ Input: []schemas.ChatMessage{{
+ Role: schemas.ChatMessageRoleUser,
+ Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hello")},
+ }},
+ Params: &schemas.ChatParameters{
+ Reasoning: &schemas.ChatReasoning{Effort: schemas.Ptr(tt.effort)},
+ },
+ })
+ require.NotNil(t, out)
+
+ if tt.wantEffort == nil {
+ if out.ChatParameters.Reasoning != nil && out.ChatParameters.Reasoning.Effort != nil {
+ t.Fatalf("expected reasoning to be stripped, got effort %q", *out.ChatParameters.Reasoning.Effort)
+ }
+ return
+ }
+ require.NotNil(t, out.ChatParameters.Reasoning, "expected reasoning to be kept")
+ require.NotNil(t, out.ChatParameters.Reasoning.Effort, "expected effort to be kept")
+ require.Equal(t, *tt.wantEffort, *out.ChatParameters.Reasoning.Effort)
+ })
+ }
+}
+
+// Alibaba Model Studio honors reasoning_effort only on qwen3.8-max and hosted
+// DeepSeek-V4 / GLM-5 series; all other models reject it (thinking is controlled
+// via enable_thinking / thinking_budget extra params there). qwen3.8-max's
+// OpenAI-compatible ladder tops out at xhigh (vendor enum: none/minimal/low/
+// medium/high/xhigh — the vendor 400'd "max" until ~2026-08-22), so the gateway
+// forwards xhigh verbatim and clamps max down to xhigh via the shared
+// normalizer.
+func TestToOpenAIChatRequest_AlibabaReasoningRouting(t *testing.T) {
+ ctx, cancel := schemas.NewBifrostContextWithCancel(nil)
+ defer cancel()
+
+ tests := []struct {
+ name string
+ model string
+ effort string
+ wantEffort *string
+ }{
+ {name: "qwen3.8-max keeps xhigh", model: "qwen3.8-max", effort: "xhigh", wantEffort: schemas.Ptr("xhigh")},
+ {name: "qwen3.8-max clamps max to xhigh", model: "qwen3.8-max", effort: "max", wantEffort: schemas.Ptr("xhigh")},
+ {name: "qwen3.8-max keeps high", model: "qwen3.8-max", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "qwen3.8-max keeps medium", model: "qwen3.8-max", effort: "medium", wantEffort: schemas.Ptr("medium")},
+ {name: "qwen3.8-max maps minimal to low", model: "qwen3.8-max", effort: "minimal", wantEffort: schemas.Ptr("low")},
+ {name: "prefixed qwen3.8-max keeps xhigh", model: "alibaba/qwen3.8-max", effort: "xhigh", wantEffort: schemas.Ptr("xhigh")},
+ {name: "prefixed qwen3.8-max clamps max to xhigh", model: "alibaba/qwen3.8-max", effort: "max", wantEffort: schemas.Ptr("xhigh")},
+ {name: "qwen3.8-max-preview keeps medium", model: "qwen3.8-max-preview", effort: "medium", wantEffort: schemas.Ptr("medium")},
+ {name: "hosted deepseek-v4-pro keeps high", model: "deepseek-v4-pro", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "hosted deepseek-v4-pro keeps max", model: "deepseek-v4-pro", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "hosted glm-5.2 keeps max", model: "glm-5.2", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "hosted glm-5.1 keeps high", model: "glm-5.1", effort: "high", wantEffort: schemas.Ptr("high")},
+ {name: "hosted glm-5.3 keeps max", model: "glm-5.3", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "hosted glm-5.5 keeps max", model: "glm-5.5", effort: "max", wantEffort: schemas.Ptr("max")},
+ {name: "qwen3.6-flash strips effort", model: "qwen3.6-flash", effort: "high", wantEffort: nil},
+ {name: "qwen-max strips effort", model: "qwen-max", effort: "low", wantEffort: nil},
+ {name: "qwen3-coder-plus strips effort", model: "qwen3-coder-plus", effort: "medium", wantEffort: nil},
+ {name: "qwen-turbo strips effort", model: "qwen-turbo", effort: "high", wantEffort: nil},
+ }
+
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ out := ToOpenAIChatRequest(ctx, &schemas.BifrostChatRequest{
+ Provider: schemas.Alibaba,
+ Model: tt.model,
+ Input: []schemas.ChatMessage{{
+ Role: schemas.ChatMessageRoleUser,
+ Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hello")},
+ }},
+ Params: &schemas.ChatParameters{
+ Reasoning: &schemas.ChatReasoning{Effort: schemas.Ptr(tt.effort)},
+ },
+ })
+ require.NotNil(t, out)
+
+ if tt.wantEffort == nil {
+ if out.ChatParameters.Reasoning != nil && out.ChatParameters.Reasoning.Effort != nil {
+ t.Fatalf("expected reasoning to be stripped, got effort %q", *out.ChatParameters.Reasoning.Effort)
+ }
+ return
+ }
+ require.NotNil(t, out.ChatParameters.Reasoning, "expected reasoning to be kept")
+ require.NotNil(t, out.ChatParameters.Reasoning.Effort, "expected effort to be kept")
+ require.Equal(t, *tt.wantEffort, *out.ChatParameters.Reasoning.Effort)
+ })
+ }
+}
diff --git a/core/providers/openai/responses_marshal_test.go b/core/providers/openai/responses_marshal_test.go
index 04292dfe3b4..1c27e459c1d 100644
--- a/core/providers/openai/responses_marshal_test.go
+++ b/core/providers/openai/responses_marshal_test.go
@@ -184,6 +184,12 @@ func TestNormalizeOpenAIReasoningEffort(t *testing.T) {
{"gpt-5.5 keeps xhigh", "gpt-5.5", "xhigh", "xhigh"},
{"gpt-5.1 downgrades max to high", "gpt-5.1", "max", "high"},
{"gpt-5.1 downgrades xhigh to high", "gpt-5.1", "xhigh", "high"},
+ {"qwen3.8-max keeps xhigh", "qwen3.8-max", "xhigh", "xhigh"},
+ {"qwen3.8-max downgrades max to xhigh", "qwen3.8-max", "max", "xhigh"},
+ {"provider-prefixed qwen3.8-max downgrades max to xhigh", "alibaba/qwen3.8-max", "max", "xhigh"},
+ {"qwen3.8-max keeps medium", "qwen3.8-max", "medium", "medium"},
+ {"qwen3.8-max downgrades minimal to low", "qwen3.8-max", "minimal", "low"},
+ {"kimi-k3 keeps max", "kimi-k3", "max", "max"},
{"standard effort passes through", "gpt-5.1", "medium", "medium"},
}
diff --git a/core/providers/openai/responses_test.go b/core/providers/openai/responses_test.go
index 318eeb6f6d7..dad5b90c380 100644
--- a/core/providers/openai/responses_test.go
+++ b/core/providers/openai/responses_test.go
@@ -613,6 +613,43 @@ func TestToOpenAIResponsesRequest_NormalizesReasoningEffort(t *testing.T) {
effort: "max",
expected: "max",
},
+ {
+ // GLM-5.3 (Z.ai) also natively supports "max" reasoning effort.
+ name: "preserves max for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "max",
+ expected: "max",
+ },
+ {
+ name: "preserves max for provider-prefixed glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "zai/glm-5.3",
+ effort: "max",
+ expected: "max",
+ },
+ {
+ // GLM-5.3 only accepts max/high/low; wider tiers are mapped.
+ name: "maps medium to high for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "medium",
+ expected: "high",
+ },
+ {
+ name: "maps none to low for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "glm-5.3",
+ effort: "none",
+ expected: "low",
+ },
+ {
+ name: "maps xhigh to max for glm-5.3",
+ provider: schemas.ModelProvider("zai"),
+ model: "zai/glm-5.3",
+ effort: "xhigh",
+ expected: "max",
+ },
}
for _, tt := range tests {
diff --git a/core/providers/openai/utils.go b/core/providers/openai/utils.go
index 5a129a2c9b0..04351aca685 100644
--- a/core/providers/openai/utils.go
+++ b/core/providers/openai/utils.go
@@ -1,6 +1,7 @@
package openai
import (
+ "strconv"
"strings"
"github.com/maximhq/bifrost/core/providers/utils"
@@ -66,6 +67,22 @@ func defaultEffortControl(model string) *schemas.EffortControl {
if acceptsMaxEffort(model) {
levels = append(levels, schemas.ReasoningEffortMax)
}
+ // GLM-5.3 narrowed reasoning_effort to exactly max/high/low and rejects
+ // every other value on the General API (low is newly added; medium/minimal/
+ // none/xhigh are gone). Publishing the narrowed ladder lets the shared
+ // normalizer clamp wider tiers onto the nearest rung — mirroring Zhipu's
+ // own Coding Plan coercion table: xhigh→max, medium→high, minimal→low —
+ // on every OpenAI-dialect mount (zhipu, zai-style custom providers).
+ // "none" is unranked in the downgrade ladder by design (it disables
+ // reasoning rather than lowering it), so it reaches "low" through Renames,
+ // which is consulted before the ladder. GLM-5.2 keeps the full legacy enum;
+ // the vendor maps the extra tiers itself.
+ if isGLM53OrLater(bareModelLower(model)) {
+ return &schemas.EffortControl{
+ Levels: []string{schemas.ReasoningEffortLow, schemas.ReasoningEffortHigh, schemas.ReasoningEffortMax},
+ Renames: map[string]string{schemas.ReasoningEffortNone: schemas.ReasoningEffortLow},
+ }
+ }
return &schemas.EffortControl{Levels: levels}
}
@@ -73,6 +90,10 @@ func defaultEffortControl(model string) *schemas.EffortControl {
// ladder is shared by every OpenAI-dialect provider, so non-OpenAI families
// supporting the tier are recognised here too — otherwise their "xhigh" is
// downgraded to "high" before the provider-specific compat pass ever runs.
+// qwen3.8-max's OpenAI-compatible enum tops out at xhigh
+// (none/minimal/low/medium/high/xhigh; the vendor 400'd "max" until ~2026-08-22),
+// so publishing xhigh here lets the shared normalizer clamp a requested "max"
+// down to xhigh instead of forwarding a value the vendor only lately tolerates.
func acceptsXHighEffort(model string) bool {
modelLower := bareModelLower(model)
if schemas.SupportsGrokXHighReasoningEffort(modelLower) {
@@ -86,7 +107,8 @@ func acceptsXHighEffort(model string) bool {
strings.Contains(modelLower, "gpt-5.3-codex") ||
strings.Contains(modelLower, "gpt-5.4") ||
strings.Contains(modelLower, "gpt-5.5") ||
- strings.Contains(modelLower, "gpt-5.6")
+ strings.Contains(modelLower, "gpt-5.6") ||
+ strings.Contains(modelLower, "qwen3.8-max")
}
// acceptsMinimalEffort reports models that natively accept "minimal" effort:
@@ -109,12 +131,69 @@ func acceptsMinimalEffort(model string) bool {
return rest == "" || strings.HasPrefix(rest, "-2")
}
-// acceptsMaxEffort reports models that natively accept "max" effort.
+// acceptsMaxEffort reports models that natively accept "max" effort
+// (e.g. GPT-5.6, DeepSeek V4, GLM-5.2+). qwen3.8-max is NOT here: its
+// OpenAI-compatible enum tops out at xhigh, and the gateway clamps a requested
+// "max" down to xhigh via the ladder published by acceptsXHighEffort.
func acceptsMaxEffort(model string) bool {
modelLower := bareModelLower(model)
return strings.Contains(modelLower, "gpt-5.6") ||
strings.Contains(modelLower, "deepseek-v4") ||
- strings.Contains(modelLower, "glm-5.2")
+ isGLM52OrLater(modelLower) ||
+ strings.Contains(modelLower, "kimi-k3") ||
+ // Kimi Code stable aliases for Kimi K3 (k3, k3-256k).
+ modelLower == "k3" || strings.HasPrefix(modelLower, "k3-")
+}
+
+// glm5Minor returns the GLM-5.x minor revision of a (lowercased) model name,
+// or -1 when the model is not a glm-5.... shape. Covers the Coding Plan
+// aliases (glm-5.2[1m]) and -air/-flash style variants via the leading-digit
+// parse. Substring-anchored so multi-segment catalog IDs
+// ("vendor/region/glm-5.3") match as well.
+func glm5Minor(modelLower string) int {
+ i := strings.Index(modelLower, "glm-5.")
+ if i < 0 {
+ return -1
+ }
+ rest := modelLower[i+len("glm-5."):]
+ digits := 0
+ for digits < len(rest) && rest[digits] >= '0' && rest[digits] <= '9' {
+ digits++
+ }
+ if digits == 0 {
+ return -1
+ }
+ minor, err := strconv.Atoi(rest[:digits])
+ if err != nil {
+ return -1
+ }
+ return minor
+}
+
+// isGLM52OrLater reports whether the (lowercased) model is a GLM-5.x revision
+// from 5.2 onward — the first Zhipu series whose OpenAI-compatible surface
+// accepts top-level reasoning_effort with the full enum including "max".
+// GLM-4.x and GLM-5.0/5.1 ignore or 400 the field (their thinking is controlled
+// via the `thinking` extra param), so the floor stays at 5.2. Matching the
+// family by version floor instead of listing each revision keeps future GLM
+// releases (glm-5.3, glm-5.5, ...) working without a gateway patch.
+func isGLM52OrLater(modelLower string) bool {
+ return glm5Minor(modelLower) >= 2
+}
+
+// isGLM53OrLater reports whether the (lowercased) model is a GLM-5.x revision
+// from 5.3 onward — the revision that narrowed reasoning_effort to exactly
+// max/high/low and made thinking non-disableable.
+func isGLM53OrLater(modelLower string) bool {
+ return glm5Minor(modelLower) >= 3
+}
+
+// isGLM53OrLaterModel reports whether the (possibly provider-prefixed) model is
+// a GLM-5.x revision from 5.3 onward — the revision that narrowed
+// reasoning_effort to exactly max/high/low.
+func isGLM53OrLaterModel(model string) bool {
+ modelLower := bareModelLower(model)
+ return isGLM53OrLater(modelLower)
}
// bareModelLower strips any provider prefix and lowercases, so the effort
@@ -126,7 +205,6 @@ func bareModelLower(model string) string {
return strings.ToLower(model)
}
-
func ConvertOpenAIMessagesToBifrostMessages(messages []OpenAIMessage) []schemas.ChatMessage {
bifrostMessages := make([]schemas.ChatMessage, len(messages))
for i, message := range messages {
diff --git a/core/providers/openai/utils_test.go b/core/providers/openai/utils_test.go
new file mode 100644
index 00000000000..7d3fec7eaba
--- /dev/null
+++ b/core/providers/openai/utils_test.go
@@ -0,0 +1,112 @@
+package openai
+
+import "testing"
+
+// glm5Minor backs the GLM version floors: isGLM52OrLater (reasoning_effort
+// support incl. "max") and isGLM53OrLaterModel (the max/high/low-only clamp).
+func TestGLM5Minor(t *testing.T) {
+ tests := []struct {
+ model string
+ want int
+ }{
+ {"glm-5", -1},
+ {"glm-5.x", -1},
+ {"glm-5.1", 1},
+ {"glm-5.2", 2},
+ {"glm-5.2[1m]", 2},
+ {"glm-5.2-air", 2},
+ {"glm-5.3", 3},
+ {"glm-5.5", 5},
+ {"glm-5.10", 10},
+ {"glm-4.7", -1},
+ {"kimi-k3", -1},
+ {"", -1},
+ }
+ for _, tt := range tests {
+ if got := glm5Minor(tt.model); got != tt.want {
+ t.Errorf("glm5Minor(%q) = %d, want %d", tt.model, got, tt.want)
+ }
+ }
+}
+
+// isGLM52OrLater is the version-floor matcher backing both
+// supportsMaxReasoningEffort and the zhipu reasoning-effort gate — it must
+// cover future GLM revisions (glm-5.3, glm-5.5, ...) and the Coding Plan
+// aliases, while staying below the 5.2 floor for GLM-5.0/5.1 and GLM-4.x.
+func TestIsGLM52OrLater(t *testing.T) {
+ tests := []struct {
+ model string
+ want bool
+ }{
+ {"glm-5.2", true},
+ {"glm-5.2[1m]", true},
+ {"glm-5.2-air", true},
+ {"glm-5.3", true},
+ {"glm-5.5", true},
+ {"glm-5.10", true},
+ {"glm-5.1", false},
+ {"glm-5", false},
+ {"glm-5.x", false},
+ {"glm-4.7", false},
+ {"glm-4.5-flash", false},
+ {"glm-4.6v", false},
+ {"kimi-k3", false},
+ {"", false},
+ }
+ for _, tt := range tests {
+ if got := isGLM52OrLater(tt.model); got != tt.want {
+ t.Errorf("isGLM52OrLater(%q) = %v, want %v", tt.model, got, tt.want)
+ }
+ }
+}
+
+// isGLM53OrLater gates the GLM-5.3 reasoning_effort clamp — 5.3 narrowed the
+// enum to max/high/low and made thinking non-disableable.
+func TestIsGLM53OrLater(t *testing.T) {
+ tests := []struct {
+ model string
+ want bool
+ }{
+ {"glm-5.3", true},
+ {"glm-5.3-air", true},
+ {"glm-5.5", true},
+ {"glm-5.2", false},
+ {"glm-5.2[1m]", false},
+ {"glm-5.1", false},
+ {"glm-5", false},
+ {"glm-4.7", false},
+ {"", false},
+ }
+ for _, tt := range tests {
+ if got := isGLM53OrLater(tt.model); got != tt.want {
+ t.Errorf("isGLM53OrLater(%q) = %v, want %v", tt.model, got, tt.want)
+ }
+ }
+}
+
+// isGLM53OrLaterModel is the prefix-aware wrapper used by the shared
+// normalizer — it strips a registered provider prefix and lowercases before
+// applying the 5.3 floor.
+func TestIsGLM53OrLaterModel(t *testing.T) {
+ tests := []struct {
+ model string
+ want bool
+ }{
+ {"glm-5.3", true},
+ {"GLM-5.3", true},
+ {"glm-5.3-air", true},
+ {"glm-5.5", true},
+ {"glm-5.2", false},
+ {"glm-5.2[1m]", false},
+ {"glm-5.1", false},
+ {"glm-5", false},
+ {"glm-4.7", false},
+ {"gpt-5.6", false},
+ {"", false},
+ }
+ for _, tt := range tests {
+ if got := isGLM53OrLaterModel(tt.model); got != tt.want {
+ t.Errorf("isGLM53OrLaterModel(%q) = %v, want %v", tt.model, got, tt.want)
+ }
+ }
+}
diff --git a/core/providers/zhipu/cachedcontents.go b/core/providers/zhipu/cachedcontents.go
new file mode 100644
index 00000000000..31c97f082fb
--- /dev/null
+++ b/core/providers/zhipu/cachedcontents.go
@@ -0,0 +1,33 @@
+package zhipu
+
+import (
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ "github.com/maximhq/bifrost/core/schemas"
+)
+
+// CachedContentCreate is unsupported on ZhipuProvider. Only Gemini and Vertex AI
+// implement the cached-content lifecycle (Google AI Studio + Vertex AI named
+// caches). Zhipu handles caching implicitly (context caching).
+func (provider *ZhipuProvider) CachedContentCreate(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCachedContentCreateRequest) (*schemas.BifrostCachedContentCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentCreateRequest, provider.GetProviderKey())
+}
+
+// CachedContentList is unsupported on ZhipuProvider (see CachedContentCreate).
+func (provider *ZhipuProvider) CachedContentList(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentListRequest) (*schemas.BifrostCachedContentListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentListRequest, provider.GetProviderKey())
+}
+
+// CachedContentRetrieve is unsupported on ZhipuProvider (see CachedContentCreate).
+func (provider *ZhipuProvider) CachedContentRetrieve(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentRetrieveRequest) (*schemas.BifrostCachedContentRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentRetrieveRequest, provider.GetProviderKey())
+}
+
+// CachedContentUpdate is unsupported on ZhipuProvider (see CachedContentCreate).
+func (provider *ZhipuProvider) CachedContentUpdate(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentUpdateRequest) (*schemas.BifrostCachedContentUpdateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentUpdateRequest, provider.GetProviderKey())
+}
+
+// CachedContentDelete is unsupported on ZhipuProvider (see CachedContentCreate).
+func (provider *ZhipuProvider) CachedContentDelete(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentDeleteRequest) (*schemas.BifrostCachedContentDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentDeleteRequest, provider.GetProviderKey())
+}
diff --git a/core/providers/zhipu/utils.go b/core/providers/zhipu/utils.go
new file mode 100644
index 00000000000..765f8bf88a9
--- /dev/null
+++ b/core/providers/zhipu/utils.go
@@ -0,0 +1,84 @@
+package zhipu
+
+import (
+ "net/url"
+ "strings"
+)
+
+const (
+ // defaultBaseURL is the Z.AI international General API (pay-as-you-go) base URL.
+ // GLM Coding Plan users override it with https://api.z.ai/api/coding/paas/v4;
+ // CN BigModel platform users with https://open.bigmodel.cn/api/paas/v4.
+ defaultBaseURL = "https://api.z.ai/api/paas/v4"
+
+ // chatCompletionsPath is the chat completions path relative to the OpenAI-compatible base URL.
+ chatCompletionsPath = "/chat/completions"
+
+ // modelsPath is the list-models path relative to the OpenAI-compatible base URL.
+ modelsPath = "/models"
+
+ // codingPlanSuffix is the Coding Plan OpenAI mount suffix.
+ codingPlanSuffix = "/coding/paas/v4"
+
+ // generalAPISuffix is the General API OpenAI mount suffix.
+ generalAPISuffix = "/paas/v4"
+
+ // anthropicMount is the Anthropic mount suffix (Coding Plan surface only).
+ anthropicMount = "/anthropic"
+
+ // anthropicMessagesPath is the messages path under the derived Anthropic mount base.
+ anthropicMessagesPath = "/v1/messages"
+)
+
+// zhipuKnownHosts are the upstream hosts whose URL shapes the suffix rewrites
+// below rely on. Custom or proxied hosts never get their paths rewritten.
+var zhipuKnownHosts = map[string]bool{
+ "api.z.ai": true, // General API + Coding Plan (international)
+ "open.bigmodel.cn": true, // BigModel platform (China)
+}
+
+// isKnownZhipuHost reports whether the base URL sits on one of Zhipu's own hosts.
+// Unparseable or scheme-less inputs are treated as custom hosts.
+func isKnownZhipuHost(base string) bool {
+ parsed, err := url.Parse(base)
+ if err != nil || parsed.Scheme == "" || parsed.Host == "" {
+ return false
+ }
+ return zhipuKnownHosts[parsed.Host]
+}
+
+// deriveAnthropicBaseURL derives the Anthropic-compatible mount base URL from the
+// configured OpenAI-compatible base URL.
+//
+// Both the General API and Coding Plan OpenAI mounts resolve to the same
+// Anthropic mount host path:
+//
+// - https://api.z.ai/api/coding/paas/v4 -> https://api.z.ai/api/anthropic
+// - https://api.z.ai/api/paas/v4 -> https://api.z.ai/api/anthropic
+//
+// (same for open.bigmodel.cn). The mount requires a Coding Plan key; requests made
+// with a General API key are rejected upstream with 401. The path rewrites only
+// apply on Zhipu's own hosts — any other base URL, including a custom or proxied
+// base that merely ends in /coding/paas/v4 or /paas/v4, keeps its configured path
+// and only gets /anthropic appended; users on exotic hosts can always create a
+// second provider instance with an explicit base URL.
+//
+// Idempotent: a base that already ends with the mount suffix is returned unchanged —
+// use_anthropic_endpoints with a base_url set to the mount itself must not append
+// /anthropic a second time.
+func deriveAnthropicBaseURL(openAIBaseURL string) string {
+ base := strings.TrimRight(openAIBaseURL, "/")
+ if strings.HasSuffix(base, anthropicMount) {
+ return base
+ }
+ if !isKnownZhipuHost(base) {
+ return base + anthropicMount
+ }
+ if strings.HasSuffix(base, codingPlanSuffix) {
+ return strings.TrimSuffix(base, codingPlanSuffix) + anthropicMount
+ }
+ if strings.HasSuffix(base, generalAPISuffix) {
+ return strings.TrimSuffix(base, generalAPISuffix) + anthropicMount
+ }
+ return base + anthropicMount
+}
diff --git a/core/providers/zhipu/utils_test.go b/core/providers/zhipu/utils_test.go
new file mode 100644
index 00000000000..0cc68a05765
--- /dev/null
+++ b/core/providers/zhipu/utils_test.go
@@ -0,0 +1,90 @@
+package zhipu
+
+import "testing"
+
+func TestDeriveAnthropicBaseURL(t *testing.T) {
+ tests := []struct {
+ name string
+ openAIBase string
+ wantMessages string // full messages URL
+ }{
+ {
+ name: "General API international",
+ openAIBase: "https://api.z.ai/api/paas/v4",
+ wantMessages: "https://api.z.ai/api/anthropic/v1/messages",
+ },
+ {
+ name: "General API China",
+ openAIBase: "https://open.bigmodel.cn/api/paas/v4",
+ wantMessages: "https://open.bigmodel.cn/api/anthropic/v1/messages",
+ },
+ {
+ name: "Coding Plan international",
+ openAIBase: "https://api.z.ai/api/coding/paas/v4",
+ wantMessages: "https://api.z.ai/api/anthropic/v1/messages",
+ },
+ {
+ name: "Coding Plan China",
+ openAIBase: "https://open.bigmodel.cn/api/coding/paas/v4",
+ wantMessages: "https://open.bigmodel.cn/api/anthropic/v1/messages",
+ },
+ {
+ name: "trailing slash is trimmed",
+ openAIBase: "https://api.z.ai/api/paas/v4/",
+ wantMessages: "https://api.z.ai/api/anthropic/v1/messages",
+ },
+ {
+ name: "custom base falls back to appending /anthropic",
+ openAIBase: "https://proxy.example.com/zhipu",
+ wantMessages: "https://proxy.example.com/zhipu/anthropic/v1/messages",
+ },
+ {
+ // Suffix semantics only hold on Zhipu's own hosts: a custom host
+ // ending in a recognized suffix must NOT have it rewritten away.
+ name: "custom base ending in /coding/paas/v4 keeps its path",
+ openAIBase: "https://proxy.example.com/api/coding/paas/v4",
+ wantMessages: "https://proxy.example.com/api/coding/paas/v4/anthropic/v1/messages",
+ },
+ {
+ name: "custom base ending in /paas/v4 keeps its path",
+ openAIBase: "https://proxy.example.com/api/paas/v4",
+ wantMessages: "https://proxy.example.com/api/paas/v4/anthropic/v1/messages",
+ },
+ {
+ // The known-host gate is exact-match: a host whose name merely
+ // contains api.z.ai is a foreign host.
+ name: "lookalike host containing api.z.ai keeps its path",
+ openAIBase: "https://api.z.ai.proxy.example.com/api/paas/v4",
+ wantMessages: "https://api.z.ai.proxy.example.com/api/paas/v4/anthropic/v1/messages",
+ },
+ {
+ // Scheme-less values must take the custom-host fallback even when
+ // the remainder parses to a known Zhipu host.
+ name: "scheme-less base takes the custom-host fallback",
+ openAIBase: "//api.z.ai/api/paas/v4",
+ wantMessages: "//api.z.ai/api/paas/v4/anthropic/v1/messages",
+ },
+ {
+ // use_anthropic_endpoints with a base_url already set to the mount
+ // itself must not append /anthropic a second time.
+ name: "Anthropic mount as base is idempotent",
+ openAIBase: "https://api.z.ai/api/anthropic",
+ wantMessages: "https://api.z.ai/api/anthropic/v1/messages",
+ },
+ {
+ name: "Anthropic mount as base with trailing slash is idempotent",
+ openAIBase: "https://open.bigmodel.cn/api/anthropic/",
+ wantMessages: "https://open.bigmodel.cn/api/anthropic/v1/messages",
+ },
+ }
+
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ base := deriveAnthropicBaseURL(tt.openAIBase)
+ full := base + anthropicMessagesPath
+ if full != tt.wantMessages {
+ t.Errorf("messages URL for %q = %q, want %q", tt.openAIBase, full, tt.wantMessages)
+ }
+ })
+ }
+}
diff --git a/core/providers/zhipu/zhipu.go b/core/providers/zhipu/zhipu.go
new file mode 100644
index 00000000000..0c58ca1f18e
--- /dev/null
+++ b/core/providers/zhipu/zhipu.go
@@ -0,0 +1,513 @@
+// Package zhipu implements the Zhipu AI (Z.AI / GLM) LLM provider.
+//
+// Zhipu runs the GLM model family on two storefronts with separate key pools
+// (api.z.ai international, open.bigmodel.cn China), each with a pay-as-you-go
+// General API surface and a GLM Coding Plan subscription surface (doc: docs/research
+// 04-provider-zhipu.md). Only the Coding Plan surface exposes the Anthropic-compatible
+// mount, selected per key/alias via use_anthropic_endpoints.
+package zhipu
+
+import (
+ "context"
+ "maps"
+ "strings"
+ "time"
+
+ "github.com/maximhq/bifrost/core/providers/anthropic"
+ "github.com/maximhq/bifrost/core/providers/openai"
+ providerUtils "github.com/maximhq/bifrost/core/providers/utils"
+ schemas "github.com/maximhq/bifrost/core/schemas"
+ "github.com/valyala/fasthttp"
+)
+
+// ZhipuProvider implements the Provider interface for Zhipu's API.
+type ZhipuProvider struct {
+ logger schemas.Logger // Logger for provider operations
+ client *fasthttp.Client // HTTP client for unary API requests (ReadTimeout bounds overall response)
+ streamingClient *fasthttp.Client // HTTP client for streaming API requests (no ReadTimeout; idle governed by NewIdleTimeoutReader)
+ networkConfig schemas.NetworkConfig // Network configuration including extra headers
+ sendBackRawRequest bool // Whether to include raw request in BifrostResponse
+ sendBackRawResponse bool // Whether to include raw response in BifrostResponse
+}
+
+// NewZhipuProvider creates a new Zhipu provider instance.
+// It initializes the HTTP client with the provided configuration and sets up response pools.
+// The client is configured with timeouts, concurrency limits, and optional proxy settings.
+func NewZhipuProvider(config *schemas.ProviderConfig, logger schemas.Logger) (*ZhipuProvider, error) {
+ config.CheckAndSetDefaults()
+
+ // Clone the NetworkConfig (including its mutable maps) so the provider never
+ // shares state with the caller's ProviderConfig — later mutations of the
+ // caller's ExtraHeaders/BetaHeaderOverrides must not affect live requests,
+ // and the BaseURL defaulting below must not write back to the caller.
+ networkConfig := config.NetworkConfig
+ networkConfig.ExtraHeaders = maps.Clone(config.NetworkConfig.ExtraHeaders)
+ networkConfig.BetaHeaderOverrides = maps.Clone(config.NetworkConfig.BetaHeaderOverrides)
+
+ requestTimeout := time.Second * time.Duration(networkConfig.DefaultRequestTimeoutInSeconds)
+ client := &fasthttp.Client{
+ ReadTimeout: requestTimeout,
+ WriteTimeout: requestTimeout,
+ MaxConnsPerHost: networkConfig.MaxConnsPerHost,
+ MaxIdleConnDuration: time.Second * time.Duration(networkConfig.KeepAliveTimeoutInSeconds),
+ MaxConnWaitTimeout: requestTimeout,
+ MaxConnDuration: time.Second * time.Duration(schemas.DefaultMaxConnDurationInSeconds),
+ ConnPoolStrategy: fasthttp.FIFO,
+ }
+
+ // Configure proxy and retry policy
+ client = providerUtils.ConfigureProxy(client, config.ProxyConfig, logger)
+ client = providerUtils.ConfigureDialer(client, networkConfig.AllowPrivateNetwork)
+ client = providerUtils.ConfigureTLS(client, networkConfig, logger)
+ streamingClient := providerUtils.BuildStreamingClient(client)
+ // Set default BaseURL if not provided
+ if networkConfig.BaseURL == "" {
+ networkConfig.BaseURL = defaultBaseURL
+ }
+ networkConfig.BaseURL = strings.TrimRight(networkConfig.BaseURL, "/")
+
+ return &ZhipuProvider{
+ logger: logger,
+ client: client,
+ streamingClient: streamingClient,
+ networkConfig: networkConfig,
+ sendBackRawRequest: config.SendBackRawRequest,
+ sendBackRawResponse: config.SendBackRawResponse,
+ }, nil
+}
+
+// bearerHeaders builds the Bearer auth headers for Zhipu's Anthropic-compatible mount
+// (Claude Code connects with an ANTHROPIC_AUTH_TOKEN, i.e. Bearer auth).
+func (provider *ZhipuProvider) bearerHeaders(key schemas.Key) map[string]string {
+ headers := map[string]string{}
+ if key.Value.GetValue() != "" {
+ headers["Authorization"] = "Bearer " + key.Value.GetValue()
+ }
+ return headers
+}
+
+// anthropicMessagesURL returns the full Anthropic-mount messages URL for this request,
+// derived from the configured OpenAI base URL and honoring per-request path overrides.
+func (provider *ZhipuProvider) anthropicMessagesURL(ctx *schemas.BifrostContext) string {
+ anthropicBase := deriveAnthropicBaseURL(provider.networkConfig.BaseURL)
+ return anthropicBase + providerUtils.GetPathFromContext(ctx, anthropicMessagesPath)
+}
+
+// GetProviderKey returns the provider identifier for Zhipu.
+func (provider *ZhipuProvider) GetProviderKey() schemas.ModelProvider {
+ return schemas.Zhipu
+}
+
+// ListModels performs a list models request to Zhipu's OpenAI-compatible API.
+func (provider *ZhipuProvider) ListModels(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) {
+ return openai.HandleOpenAIListModelsRequest(
+ ctx,
+ provider.client,
+ request,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, modelsPath),
+ keys,
+ provider.networkConfig.ExtraHeaders,
+ provider.GetProviderKey(),
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ )
+}
+
+// TextCompletion is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionRequest, provider.GetProviderKey())
+}
+
+// TextCompletionStream is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionStreamRequest, provider.GetProviderKey())
+}
+
+// ChatCompletion performs a chat completion request to Zhipu's API.
+func (provider *ZhipuProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ return anthropic.HandleAnthropicChatCompletionRequest(
+ ctx,
+ provider.client,
+ provider.anthropicMessagesURL(ctx),
+ request,
+ anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Zhipu,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ },
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ nil,
+ provider.logger,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIChatCompletionRequest(
+ ctx,
+ provider.client,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, chatCompletionsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ )
+}
+
+// ChatCompletionStream performs a streaming chat completion request to Zhipu's API.
+// It supports real-time streaming of responses using Server-Sent Events (SSE).
+// Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails.
+func (provider *ZhipuProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ jsonData, bifrostErr := anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Zhipu,
+ IsStreaming: true,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ })
+ if bifrostErr != nil {
+ return nil, bifrostErr
+ }
+
+ return anthropic.HandleAnthropicChatCompletionStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.anthropicMessagesURL(ctx),
+ jsonData,
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ provider.networkConfig.BetaHeaderOverrides,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ schemas.Zhipu,
+ postHookRunner,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true)
+ return openai.HandleOpenAIChatCompletionStreaming(
+ ctx,
+ provider.streamingClient,
+ provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, chatCompletionsPath),
+ request,
+ openai.BearerAuthHeader(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ schemas.Zhipu,
+ postHookRunner,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+}
+
+// Responses performs a Responses API request against Zhipu's Anthropic-compatible
+// mount when use_anthropic_endpoints is set (GLM Coding Plan only), and otherwise
+// falls back to chat completions on the OpenAI-compatible mount (api.z.ai exposes
+// no /responses endpoint).
+func (provider *ZhipuProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ return anthropic.HandleAnthropicResponsesRequest(
+ ctx,
+ provider.client,
+ provider.anthropicMessagesURL(ctx),
+ request,
+ anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Zhipu,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ },
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ nil,
+ provider.logger,
+ )
+ }
+
+ chatResponse, err := provider.ChatCompletion(ctx, key, request.ToChatRequest())
+ if err != nil {
+ return nil, err
+ }
+
+ return chatResponse.ToBifrostResponsesResponse(), nil
+}
+
+// ResponsesStream performs a streaming Responses API request against Zhipu's
+// Anthropic-compatible mount when use_anthropic_endpoints is set, and otherwise
+// falls back to streaming chat completions on the OpenAI-compatible mount.
+func (provider *ZhipuProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ if anthropic.ResolveUseAnthropicEndpoints(ctx, key) {
+ jsonData, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{
+ Provider: schemas.Zhipu,
+ IsStreaming: true,
+ ShouldSendBackRawRequest: provider.sendBackRawRequest,
+ ShouldSendBackRawResponse: provider.sendBackRawResponse,
+ })
+ if bifrostErr != nil {
+ return nil, bifrostErr
+ }
+
+ return anthropic.HandleAnthropicResponsesStream(
+ ctx,
+ provider.streamingClient,
+ provider.anthropicMessagesURL(ctx),
+ jsonData,
+ provider.bearerHeaders(key),
+ provider.networkConfig.ExtraHeaders,
+ provider.networkConfig.StreamIdleTimeoutInSeconds,
+ provider.networkConfig.BetaHeaderOverrides,
+ providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
+ providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
+ provider.GetProviderKey(),
+ postHookRunner,
+ nil,
+ nil,
+ provider.logger,
+ postHookSpanFinalizer,
+ )
+ }
+
+ ctx.SetValue(schemas.BifrostContextKeyIsResponsesToChatCompletionFallback, true)
+ return provider.ChatCompletionStream(
+ ctx,
+ postHookRunner,
+ postHookSpanFinalizer,
+ key,
+ request.ToChatRequest(),
+ )
+}
+
+// Embedding is not supported by the Zhipu provider (not offered on the api.z.ai
+// international endpoint; CN BigModel platform only).
+func (provider *ZhipuProvider) Embedding(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.EmbeddingRequest, provider.GetProviderKey())
+}
+
+// Speech is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) Speech(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostSpeechRequest) (*schemas.BifrostSpeechResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechRequest, provider.GetProviderKey())
+}
+
+// Rerank is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostRerankRequest) (*schemas.BifrostRerankResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.RerankRequest, provider.GetProviderKey())
+}
+
+// OCR is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) OCR(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostOCRRequest) (*schemas.BifrostOCRResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.OCRRequest, provider.GetProviderKey())
+}
+
+// SpeechStream is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) SpeechStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostSpeechRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechStreamRequest, provider.GetProviderKey())
+}
+
+// Transcription is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) Transcription(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTranscriptionRequest) (*schemas.BifrostTranscriptionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionRequest, provider.GetProviderKey())
+}
+
+// TranscriptionStream is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) TranscriptionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTranscriptionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionStreamRequest, provider.GetProviderKey())
+}
+
+// ImageGeneration is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ImageGeneration(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageGenerationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationRequest, provider.GetProviderKey())
+}
+
+// ImageGenerationStream is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ImageGenerationStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageGenerationRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationStreamRequest, provider.GetProviderKey())
+}
+
+// ImageEdit is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ImageEdit(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageEditRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditRequest, provider.GetProviderKey())
+}
+
+// ImageEditStream is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ImageEditStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageEditRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditStreamRequest, provider.GetProviderKey())
+}
+
+// ImageVariation is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ImageVariation(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageVariationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageVariationRequest, provider.GetProviderKey())
+}
+
+// VideoGeneration is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoGeneration(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoGenerationRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoGenerationRequest, provider.GetProviderKey())
+}
+
+// VideoRetrieve is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoRetrieve(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRetrieveRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRetrieveRequest, provider.GetProviderKey())
+}
+
+// VideoDownload is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoDownload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDownloadRequest) (*schemas.BifrostVideoDownloadResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDownloadRequest, provider.GetProviderKey())
+}
+
+// VideoDelete is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoDelete(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDeleteRequest) (*schemas.BifrostVideoDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDeleteRequest, provider.GetProviderKey())
+}
+
+// VideoList is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoList(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoListRequest) (*schemas.BifrostVideoListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoListRequest, provider.GetProviderKey())
+}
+
+// VideoEdit is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoEdit(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoEditRequest) (*schemas.BifrostVideoEditResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoEditRequest, provider.GetProviderKey())
+}
+
+// VideoRemix is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) VideoRemix(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRemixRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRemixRequest, provider.GetProviderKey())
+}
+
+// FileUpload is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) FileUpload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostFileUploadRequest) (*schemas.BifrostFileUploadResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileUploadRequest, provider.GetProviderKey())
+}
+
+// FileList is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) FileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileListRequest) (*schemas.BifrostFileListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileListRequest, provider.GetProviderKey())
+}
+
+// FileRetrieve is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) FileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileRetrieveRequest) (*schemas.BifrostFileRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileRetrieveRequest, provider.GetProviderKey())
+}
+
+// FileDelete is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) FileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileDeleteRequest) (*schemas.BifrostFileDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileDeleteRequest, provider.GetProviderKey())
+}
+
+// FileContent is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) FileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileContentRequest) (*schemas.BifrostFileContentResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.FileContentRequest, provider.GetProviderKey())
+}
+
+// BatchCreate is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) BatchCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostBatchCreateRequest) (*schemas.BifrostBatchCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCreateRequest, provider.GetProviderKey())
+}
+
+// BatchList is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) BatchList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchListRequest) (*schemas.BifrostBatchListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchListRequest, provider.GetProviderKey())
+}
+
+// BatchRetrieve is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) BatchRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchRetrieveRequest) (*schemas.BifrostBatchRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchRetrieveRequest, provider.GetProviderKey())
+}
+
+// BatchCancel is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) BatchCancel(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchCancelRequest) (*schemas.BifrostBatchCancelResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCancelRequest, provider.GetProviderKey())
+}
+
+// BatchDelete is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) BatchDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchDeleteRequest) (*schemas.BifrostBatchDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchDeleteRequest, provider.GetProviderKey())
+}
+
+// BatchResults is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) BatchResults(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchResultsRequest) (*schemas.BifrostBatchResultsResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchResultsRequest, provider.GetProviderKey())
+}
+
+// CountTokens is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) CountTokens(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostCountTokensResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CountTokensRequest, provider.GetProviderKey())
+}
+
+// Compaction is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) Compaction(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCompactionRequest) (*schemas.BifrostCompactionResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.CompactionRequest, provider.GetProviderKey())
+}
+
+// ContainerCreate is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerCreateRequest) (*schemas.BifrostContainerCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerCreateRequest, provider.GetProviderKey())
+}
+
+// ContainerList is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerListRequest) (*schemas.BifrostContainerListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerListRequest, provider.GetProviderKey())
+}
+
+// ContainerRetrieve is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerRetrieveRequest) (*schemas.BifrostContainerRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerRetrieveRequest, provider.GetProviderKey())
+}
+
+// ContainerDelete is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerDeleteRequest) (*schemas.BifrostContainerDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerDeleteRequest, provider.GetProviderKey())
+}
+
+// ContainerFileCreate is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerFileCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerFileCreateRequest) (*schemas.BifrostContainerFileCreateResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileCreateRequest, provider.GetProviderKey())
+}
+
+// ContainerFileList is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerFileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileListRequest) (*schemas.BifrostContainerFileListResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileListRequest, provider.GetProviderKey())
+}
+
+// ContainerFileRetrieve is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerFileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileRetrieveRequest) (*schemas.BifrostContainerFileRetrieveResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileRetrieveRequest, provider.GetProviderKey())
+}
+
+// ContainerFileContent is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerFileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileContentRequest) (*schemas.BifrostContainerFileContentResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileContentRequest, provider.GetProviderKey())
+}
+
+// ContainerFileDelete is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) ContainerFileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileDeleteRequest) (*schemas.BifrostContainerFileDeleteResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileDeleteRequest, provider.GetProviderKey())
+}
+
+// Passthrough is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) Passthrough(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (*schemas.BifrostPassthroughResponse, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughRequest, provider.GetProviderKey())
+}
+
+// PassthroughStream is not supported by the Zhipu provider.
+func (provider *ZhipuProvider) PassthroughStream(_ *schemas.BifrostContext, _ schemas.PostHookRunner, _ func(context.Context), _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
+ return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughStreamRequest, provider.GetProviderKey())
+}
diff --git a/core/providers/zhipu/zhipu_test.go b/core/providers/zhipu/zhipu_test.go
new file mode 100644
index 00000000000..28d225900aa
--- /dev/null
+++ b/core/providers/zhipu/zhipu_test.go
@@ -0,0 +1,61 @@
+package zhipu_test
+
+import (
+ "os"
+ "strings"
+ "testing"
+
+ "github.com/maximhq/bifrost/core/internal/llmtests"
+
+ "github.com/maximhq/bifrost/core/schemas"
+)
+
+func TestZhipu(t *testing.T) {
+ t.Parallel()
+ if strings.TrimSpace(os.Getenv("ZHIPU_API_KEY")) == "" {
+ t.Skip("Skipping Zhipu tests because ZHIPU_API_KEY is not set")
+ }
+
+ client, ctx, cancel, err := llmtests.SetupTest()
+ if err != nil {
+ t.Fatalf("Error initializing test setup: %v", err)
+ }
+ defer cancel()
+ defer client.Shutdown()
+
+ testConfig := llmtests.ComprehensiveTestConfig{
+ Provider: schemas.Zhipu,
+ ChatModel: "glm-4.7-flash", // free tier
+ Fallbacks: []schemas.Fallback{
+ {Provider: schemas.Zhipu, Model: "glm-4.7-flash"},
+ {Provider: schemas.Zhipu, Model: "glm-4.5-flash"},
+ },
+ TextModel: "glm-4.7-flash",
+ EmbeddingModel: "", // Zhipu embeddings are not offered on api.z.ai intl
+ ReasoningModel: "glm-5.2",
+ Scenarios: llmtests.TestScenarios{
+ SimpleChat: true,
+ CompletionStream: true,
+ MultiTurnConversation: true,
+ ToolCalls: true,
+ ToolCallsStreaming: true,
+ End2EndToolCalling: true,
+ AutomaticFunctionCall: true,
+ ImageURL: false,
+ ImageBase64: false,
+ MultipleImages: false,
+ CompleteEnd2End: true,
+ Embedding: false,
+ ListModels: true,
+ // The Reasoning scenario runs via the Responses API, which Zhipu does
+ // not expose — reasoning_effort routing is covered by unit tests and
+ // dogfood smoke tests instead.
+ Reasoning: false,
+ PassThroughExtraParams: true,
+ },
+ }
+
+ t.Run("ZhipuTests", func(t *testing.T) {
+ llmtests.RunAllComprehensiveTests(t, client, ctx, testConfig)
+ })
+}
diff --git a/core/schemas/bifrost.go b/core/schemas/bifrost.go
index d559c9d6617..1a5a169600b 100644
--- a/core/schemas/bifrost.go
+++ b/core/schemas/bifrost.go
@@ -73,6 +73,9 @@ const (
Fireworks ModelProvider = "fireworks"
Sarvam ModelProvider = "sarvam"
Wafer ModelProvider = "wafer"
+ Alibaba ModelProvider = "alibaba"
+ Kimi ModelProvider = "kimi"
+ Zhipu ModelProvider = "zhipu"
)
// SupportedBaseProviders is the list of base providers allowed for custom providers.
@@ -118,6 +121,9 @@ var StandardProviders = []ModelProvider{
Fireworks,
Sarvam,
Wafer,
+ Alibaba,
+ Kimi,
+ Zhipu,
}
// RequestType represents the type of request being made to a provider.
diff --git a/core/utils.go b/core/utils.go
index 148650256a0..477f290a599 100644
--- a/core/utils.go
+++ b/core/utils.go
@@ -87,6 +87,7 @@ var rateLimitPatterns = []string{
// dynamicallyConfigurableProviders is the list of providers that can be dynamically configured.
// Excluding providers that require extra configuration (e.g. Ollama, SGL, vLLM).
var dynamicallyConfigurableProviders = []schemas.ModelProvider{
+ schemas.Alibaba,
schemas.Anthropic,
schemas.Azure,
schemas.Bedrock,
@@ -98,6 +99,7 @@ var dynamicallyConfigurableProviders = []schemas.ModelProvider{
schemas.Gemini,
schemas.Groq,
schemas.HuggingFace,
+ schemas.Kimi,
schemas.Mistral,
schemas.Nebius,
schemas.OpenAI,
@@ -108,6 +110,7 @@ var dynamicallyConfigurableProviders = []schemas.ModelProvider{
schemas.Vertex,
schemas.Wafer,
schemas.XAI,
+ schemas.Zhipu,
}
// isModelRequired returns true if the request type requires a model
diff --git a/docs/docs.json b/docs/docs.json
index 26d5a6ea051..f2dcfc72ce8 100644
--- a/docs/docs.json
+++ b/docs/docs.json
@@ -125,6 +125,7 @@
"icon": "bars-progress",
"pages": [
"providers/supported-providers/overview",
+ "providers/supported-providers/alibaba",
"providers/supported-providers/anthropic",
"providers/supported-providers/azure",
"providers/supported-providers/azure-model-router",
@@ -139,6 +140,7 @@
"providers/supported-providers/gemini",
"providers/supported-providers/groq",
"providers/supported-providers/huggingface",
+ "providers/supported-providers/kimi",
"providers/supported-providers/mistral",
"providers/supported-providers/nebius",
"providers/supported-providers/ollama",
@@ -155,7 +157,8 @@
"providers/supported-providers/vertex",
"providers/supported-providers/vllm",
"providers/supported-providers/wafer",
- "providers/supported-providers/xai"
+ "providers/supported-providers/xai",
+ "providers/supported-providers/zhipu"
]
},
"providers/routing-rules",
diff --git a/docs/providers/supported-providers/alibaba.mdx b/docs/providers/supported-providers/alibaba.mdx
new file mode 100644
index 00000000000..c95dfe696b3
--- /dev/null
+++ b/docs/providers/supported-providers/alibaba.mdx
@@ -0,0 +1,307 @@
+---
+title: "Alibaba Cloud Model Studio"
+description: "Alibaba Cloud Model Studio (Qwen / DashScope) conversion guide - OpenAI-compatible chat, responses, and embeddings by default, optional per-key/per-alias Anthropic-compatible endpoints, streaming, tool calling, and reasoning"
+icon: "a"
+---
+
+## Overview
+
+Alibaba Cloud Model Studio (Bailian / DashScope) is a provider with a dedicated Bifrost provider implementation. By default, Chat Completions, the Responses API, and Embeddings use Model Studio's **OpenAI-compatible** endpoints. Each key (or an individual alias) can opt into routing Chat Completions and the Responses API through Model Studio's **Anthropic-compatible** endpoint instead, using the `use_anthropic_endpoints` toggle. Key characteristics:
+
+- **OpenAI-compatible by default** - Chat Completions use `/chat/completions`, the Responses API uses `/responses` (supported models), authenticated with a bearer token
+- **Optional Anthropic-compatible mode** - Set `use_anthropic_endpoints` on a key (or override it per-alias) to route Chat Completions and the Responses API through the Anthropic-compatible Messages endpoint, authenticated with `x-api-key`, using the shared Anthropic request/response converters
+- **Streaming support** - Server-Sent Events for chat and responses, in both endpoint modes
+- **Tool calling** - Function tools are supported on both the OpenAI-compatible and Anthropic-compatible paths
+- **Reasoning support** - `enable_thinking` / `thinking_budget` extra params and `reasoning_effort` (supported models) on the OpenAI-compatible path; Anthropic `thinking` conversion in Anthropic-compatible mode
+- **Embeddings** - `text-embedding-v4`/`v3` via the OpenAI-compatible `/embeddings` endpoint (unaffected by `use_anthropic_endpoints`)
+
+### Supported Operations
+
+| Operation | Non-Streaming | Streaming | Endpoint (default) | Endpoint (`use_anthropic_endpoints: true`) |
+|-----------|---------------|-----------|---------------------|---------------------------------------------|
+| Chat Completions | ✅ | ✅ | `/chat/completions` | Anthropic Messages endpoint |
+| Responses API | ✅ | ✅ | `/responses` (supported models) | Anthropic Messages endpoint |
+| Text Completions | ❌ | ❌ | - | - |
+| List Models | ✅ | - | `/models` | `/models` (unaffected) |
+| Embeddings | ✅ | - | `/embeddings` | `/embeddings` (unaffected) |
+| Image Generation | ❌ | ❌ | - | - |
+| Speech (TTS) | ❌ | ❌ | - | - |
+| Transcriptions (STT) | ❌ | ❌ | - | - |
+| Files | ❌ | ❌ | - | - |
+| Batch | ❌ | ❌ | - | - |
+
+
+**Unsupported Operations** (❌): Text Completions, Image Generation, Speech, Transcriptions, Files, Batch, cached content, containers, token counting, compaction, OCR, rerank, video, and passthrough are not supported through this provider. These return `UnsupportedOperationError`.
+
+
+## Setup & Configuration
+
+Configure Alibaba Cloud Model Studio as a provider.
+
+
+
+
+1. Navigate to **Models** > **Model Providers**. Look for **Alibaba** under **Configured Providers**. If it is missing, click on **Add New Provider** and select **Alibaba**.
+2. Click **Add Key** or edit an existing key.
+3. Set a name for your key.
+4. Paste your API key directly or use an environment variable (for example, `env.ALIBABA_API_KEY`).
+5. Set **Allowed Models** to **All Models** (default) or the specific model allowlist you want this key to serve.
+6. Leave **Use Anthropic Endpoints** off to use the OpenAI-compatible endpoints (the default), or turn it on to route Chat Completions and the Responses API through the Anthropic-compatible endpoint instead. See [Anthropic-Compatible Endpoints](#anthropic-compatible-endpoints-optional) below.
+7. Save the provider configuration.
+
+
+
+
+```json
+{
+ "providers": {
+ "alibaba": {
+ "keys": [
+ {
+ "name": "alibaba-key-1",
+ "value": "env.ALIBABA_API_KEY",
+ "models": [
+ "*"
+ ],
+ "weight": 1.0
+ }
+ ]
+ }
+ }
+}
+```
+
+
+
+Refer to the API documentation for [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider).
+
+
+
+```go
+case schemas.Alibaba:
+ return []schemas.Key{{
+ Name: "alibaba-key-1",
+ Value: *schemas.NewSecretVar("env.ALIBABA_API_KEY"),
+ Models: []string{"*"},
+ Weight: 1.0,
+ }}, nil
+```
+
+
+
+
+### Base URL (region & plan)
+
+Keys, regions, and billing modes are **paired** — a key only works against the host it was issued for, and mismatches return 401. The default base URL is the international (Singapore) pay-as-you-go host. Override `network_config.base_url` for other regions, workspace-dedicated hosts, or subscription plans:
+
+| Surface | `base_url` |
+|---------|------------|
+| Pay-as-you-go, Singapore (default) | `https://dashscope-intl.aliyuncs.com/compatible-mode/v1` |
+| Pay-as-you-go, China (Beijing) | `https://dashscope.aliyuncs.com/compatible-mode/v1` |
+| Pay-as-you-go, US (Virginia) | `https://dashscope-us.aliyuncs.com/compatible-mode/v1` |
+| Workspace-dedicated host | `https://{WorkspaceId}.{region}.maas.aliyuncs.com/compatible-mode/v1` |
+| Token Plan (Singapore) | `https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1` |
+
+The Anthropic-compatible base URL is derived automatically from the configured OpenAI-compatible base URL. The derivation is idempotent: a `base_url` already ending in `/apps/anthropic` (the mount itself) is used as-is rather than getting the suffix appended a second time.
+
+
+Subscription plans (Token Plan, Coding Plan) are billed for interactive AI coding/agent tools and use an **exact, case-sensitive model allowlist** — model IDs must match the plan catalog exactly or the request 404s. Routing general gateway traffic through plan endpoints may breach the vendor's terms of service.
+
+
+---
+
+## Anthropic-Compatible Endpoints (optional)
+
+Model Studio exposes an Anthropic-compatible Messages endpoint alongside its default OpenAI-compatible API. Setting `use_anthropic_endpoints` routes Chat Completions and the Responses API through that endpoint instead — List Models and Embeddings are unaffected and always use the OpenAI-compatible endpoints.
+
+The setting can be configured per key, and overridden per model alias:
+
+- **Key-level** - Sets the default endpoint mode for every request made with that key.
+- **Alias-level** - Overrides the key-level default for a single alias, so one key can serve some aliases through the OpenAI-compatible endpoints and others through the Anthropic-compatible endpoint.
+
+If neither is set, requests fall back to the OpenAI-compatible endpoints.
+
+**Parameter surface on the Anthropic mount** (empirical contract, live-verified 2026-08-23 — the mount's own API page documents no `output_config.effort` field; the mount honors it as a passthrough to its OpenAI-dialect backend, whose per-model `reasoning_effort` ladder is documented on the Model Studio model page):
+
+- `output_config.effort` is forwarded for **qwen3.8-max**, **hosted glm-5.2+**, and **deepseek-v4-pro/flash**. The mount does not map out-of-enum values server-side — it validates the field against each model family's `reasoning_effort` enum and rejects anything else — so the gateway emits only officially valid values per family (matrix from the Model Studio model-page docs, supplied 2026-08-23): **qwen3.8-max** accepts `xhigh`/`medium`/`low` (a requested `max` is clamped to `xhigh`; every other value forwards verbatim); **glm-5.3+** accepts `max`/`high`/`low` (`xhigh`→`max`, `medium`→`high`, `minimal`/`none`→`low`); **glm-5.2/glm-5.1/glm-5** and non-dated **deepseek-v4-pro/flash** accept `high`/`max` (`xhigh`→`max`, `low`/`medium`/`minimal`/`none`→`high` — `low` is out of this family's enum, so the mildest tiers collapse onto the mildest valid value); the dated snapshots **deepseek-v4-pro-0813**/**deepseek-v4-flash-0731** accept `max`/`high`/`low` (`xhigh`/`medium`→`high`, `minimal`/`none`→`low`). The mount also rejects `output_config.effort` and `thinking.budget_tokens` set together (`'reasoning_effort' and 'thinking_budget' cannot be set simultaneously`) and engages thinking on its own from the effort value, so when an effort is set Bifrost sends the effort alone and omits the `thinking` field. (Verified live 2026-08-23.) For all other models the field is stripped.
+- `thinking: {"type": "enabled"|"disabled", "budget_tokens": N}`, `output_config.format` (JSON Schema structured output), `cache_control` (ephemeral prompt caching), image/video content blocks, `tools`, `tool_choice` (`auto`/`any`/`none`/`tool`), `top_p`, `top_k`, `stop_sequences`, and `metadata.user_id` pass through. Note the mount's `temperature` range is **[0, 2)** and its `max_tokens` semantics differ per model family (qwen3.8-max and deepseek-v4-* count reply + chain-of-thought together).
+- Anthropic-only extensions (`speed`, `inference_geo`, `service_tier`, `container`, `mcp_servers`, `context_management`, `output_config.task_budget`) are stripped — the mount does not accept them.
+
+
+
+
+On the key form, toggle **Use Anthropic Endpoints** (off by default). To override this for a specific alias, open that alias's expanded row in the deployments table and toggle **Use Anthropic endpoints** under **Alibaba overrides** — this takes priority over the key-level setting for that alias only.
+
+
+
+The `use_anthropic_endpoints` boolean is part of the same key payload used by [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider), and the alias payload for that key's `models` entries.
+
+
+
+```json
+{
+ "providers": {
+ "alibaba": {
+ "keys": [
+ {
+ "name": "alibaba-key-1",
+ "value": "env.ALIBABA_API_KEY",
+ "models": [
+ "*"
+ ],
+ "weight": 1.0,
+ "use_anthropic_endpoints": true
+ }
+ ]
+ }
+ }
+}
+```
+
+To override this per-alias (for example, on a virtual key's model config), set `use_anthropic_endpoints` alongside the alias's `model_id`:
+
+```json
+{
+ "model_id": "qwen3.8-max",
+ "use_anthropic_endpoints": false
+}
+```
+
+| Field | Type | Required | Description |
+|-------|------|----------|--------------|
+| `use_anthropic_endpoints` | boolean | No | Routes chat completions and responses requests through Anthropic-compatible endpoints. Default: `false`. |
+
+
+
+
+---
+
+# 1. Chat Completions
+
+## Request Parameters
+
+By default, Chat Completions use Model Studio's OpenAI-compatible `/chat/completions` endpoint, authenticated with `Authorization: Bearer `. For the full parameter reference and message conversion behavior, see [OpenAI Chat Completions](/providers/supported-providers/openai#1-chat-completions).
+
+When `use_anthropic_endpoints` is enabled, requests are sent instead to the Anthropic-compatible Messages endpoint, authenticated with `x-api-key: `, and built using the shared Anthropic converters. For that parameter reference and message conversion behavior, see [Anthropic Chat Completions](/providers/supported-providers/anthropic#1-chat-completions).
+
+### Authentication
+
+| Mode | Header |
+|------|--------|
+| Default (OpenAI-compatible) | `Authorization: Bearer ` |
+| `use_anthropic_endpoints: true` | `x-api-key: ` |
+
+Bifrost sets the correct header automatically based on the resolved endpoint mode for the request.
+
+### Reasoning Parameter
+
+- **Default (OpenAI-compatible):** `reasoning_effort` is forwarded only for the models that accept it (qwen3.8-max, hosted DeepSeek-V4 series, hosted GLM-5 series) and stripped for all others — thinking on the remaining hybrid models is controlled via the `enable_thinking` / `thinking_budget` extra params instead. Valid values and mappings follow the official Model Studio `reasoning_effort` matrix (vendor API docs):
+ - **qwen3.8-max** — default `xhigh`; valid `xhigh`/`medium`/`low`. `max` and `high` map to `xhigh`, `minimal` maps to `low`, and `none` disables thinking (`enable_thinking: false`). `reasoning_effort` and `thinking_budget` cannot both be set. The gateway clamps `max`→`xhigh` itself (the vendor 400'd `max` until ~2026-08-22, so it is never forwarded) and maps `minimal`→`low`.
+ - **DeepSeek-V4 and GLM series** (`glm-5.2`, `glm-5.1`, `glm-5`, `deepseek-v4-pro`, `deepseek-v4-flash` — excluding dated snapshots) — default `high`; valid `high`/`max`. `low`, `medium`, `minimal`, and `none` map to `high` (the mildest valid tier); `xhigh` maps to `max`.
+ - **`deepseek-v4-flash-0731` and `deepseek-v4-pro-0813`** (dated snapshots) — default `high`; valid `max`/`high`/`low`. `medium` maps to `high`; `xhigh` maps to `high`.
+ - **GLM-5.3 and later revisions** — part of the hosted GLM-5 series forward set; default `max`, valid `max`/`high`/`low`. `xhigh` maps to `max`, `medium` to `high`, `minimal`/`none` to `low`. The model always performs thinking (`enable_thinking` only `true`).
+ - **kimi-k3** (supplied by Alibaba Cloud) — default `max`; valid `max`/`high`/`low`. The model always performs thinking (`enable_thinking` only `true`). Effort is still stripped for it on this mount (outside the forward set), so its thinking runs at the model default.
+ Note that forced `tool_choice` (`"required"` or a pinned function) is rejected by the vendor while thinking is enabled on hybrid models.
+- **`use_anthropic_endpoints: true`:** Reasoning/thinking parameters are mapped through the Anthropic converters (`reasoning` → `thinking`), the same as the [Anthropic provider](/providers/supported-providers/anthropic#1-chat-completions), with one mount-specific rule: an effort value travels alone as `output_config.effort` (clamped per model family to the officially valid values listed in the [Anthropic-Compatible Endpoints](#anthropic-compatible-endpoints-optional) section) with no synthesized `thinking` field — the mount rejects the two together and engages thinking itself from the effort value.
+
+### Extra Parameters
+
+Model Studio enables passthrough extra parameters for Chat Completions and the Responses API on the default OpenAI-compatible endpoints. Vendor-specific body params such as `enable_thinking`, `thinking_budget`, or `clear_thinking` can be sent under `extra_params` with the `x-bf-passthrough-extra-params: true` header — the header is what extracts `extra_params` from the request body at ingress, and it is required on either endpoint mode. Once extracted, the params are merged into the outbound request body by default on the OpenAI-compatible endpoints; the Anthropic-compatible path performs no such outbound merge, so extra parameters are effectively dropped when `use_anthropic_endpoints` is enabled.
+
+---
+
+# 2. Responses API
+
+- **Default (OpenAI-compatible):** Responses requests use Model Studio's native OpenAI-compatible `/responses` endpoint (supported models only — the hybrid-thinking Qwen3 series, hosted DeepSeek-V4, and hosted GLM models).
+- **`use_anthropic_endpoints: true`:** Responses requests are sent natively to the Anthropic-compatible Messages endpoint — both non-streaming and streaming requests build an Anthropic-format request body directly and convert the response back to Bifrost's Responses format.
+
+
+Model Studio's `/responses` mount cannot continue a tool-call conversation: it rejects tool-result items in the input. Multi-step tool calling loops should use Chat Completions, which fully support tool-result messages.
+
+
+---
+
+# 3. List Models
+
+Lists available models from Model Studio through the OpenAI-compatible `/models` endpoint. The Anthropic-compatible mount has no models endpoint (404 there), so List Models always uses the OpenAI-compatible mount regardless of `use_anthropic_endpoints`.
+
+---
+
+# 4. Embeddings
+
+Text embeddings through the OpenAI-compatible `/embeddings` endpoint (`text-embedding-v4`, `text-embedding-v3`). Unaffected by `use_anthropic_endpoints`.
+
+---
+
+## Unsupported Features
+
+| Feature | Reason |
+|---------|--------|
+| Text Completions | No `/completions` endpoint on Model Studio's OpenAI-compatible mount |
+| Image Generation | Uses separate Model Studio APIs, not the chat surface |
+| Speech/TTS | Not offered through this provider |
+| Transcription/STT | Not offered through this provider |
+| Batch Operations | Not implemented through this provider |
+| File Management | Not offered through this provider |
+| Cached Content | Only Gemini and Vertex AI support cached content in Bifrost |
+| Container Management | Not offered by Model Studio through this provider |
+| Token Counting | Not offered through this provider |
+| Rerank/OCR/Video | Not offered through this provider |
+
+---
+
+## Caveats
+
+
+**Severity**: Low
+**Behavior**: Alibaba defaults to `https://dashscope-intl.aliyuncs.com/compatible-mode/v1` (international pay-as-you-go)
+**Impact**: Keys issued for other regions, workspace-dedicated hosts, or subscription plans must override `network_config.base_url`; a key/host mismatch returns 401
+**Code**: `NewAlibabaProvider` sets the default base URL when no provider-level base URL is configured
+
+
+
+**Severity**: Low
+**Behavior**: The Anthropic-compatible base URL is derived from the configured OpenAI-compatible base URL (legacy hosts, workspace-dedicated hosts, and plan hosts are all recognized; a base already ending in the mount suffix is returned unchanged)
+**Impact**: A completely custom OpenAI-compatible host that matches none of the known shapes may not derive a valid Anthropic mount URL
+**Code**: `deriveAnthropicBaseURL` in `core/providers/alibaba/utils.go`
+
+
+
+**Severity**: Low
+**Behavior**: An alias-level `use_anthropic_endpoints` override always wins over the key-level setting for that alias; if neither is set, requests default to the OpenAI-compatible endpoints
+**Impact**: A single key can serve some aliases through OpenAI-compatible endpoints and others through the Anthropic-compatible endpoint
+**Code**: `anthropic.ResolveUseAnthropicEndpoints` in `core/providers/anthropic/utils.go`, used by `ChatCompletion`, `ChatCompletionStream`, `Responses`, and `ResponsesStream` in `core/providers/alibaba/alibaba.go`
+
+
+
+**Severity**: Medium
+**Behavior**: On the OpenAI-compatible path, `reasoning_effort` is only forwarded for qwen3.8-max and the hosted DeepSeek-V4 / GLM-5 series; it is stripped for every other model (the vendor rejects the field there)
+**Impact**: Thinking on hybrid Qwen3 models must be controlled via `enable_thinking` / `thinking_budget` extra params; a `reasoning_effort` set on an unsupported model is silently dropped
+**Code**: `applyAlibabaReasoning` in `core/providers/openai/chat.go`
+
+
+
+**Severity**: Medium
+**Behavior**: Model Studio's `/responses` mount rejects tool-result items in the input (`function_call_output` / tool messages), so multi-step tool calling loops fail with a vendor 400
+**Impact**: End-to-end tool calling works on Chat Completions only; Responses requests with tool results in the input return a provider error
+
+
+
+**Severity**: Medium
+**Behavior**: Hybrid Qwen3 models (e.g. qwen3.6-flash) run with thinking enabled by default, and reasoning tokens count against `max_completion_tokens` — a tight cap can be consumed entirely by reasoning, yielding empty content with `finish_reason: length`
+**Impact**: Requests with small token caps should either raise the cap or disable thinking via `extra_params: {"enable_thinking": false}`
+
+
+
+**Severity**: Low
+**Behavior**: The `x-bf-passthrough-extra-params: true` header is still required to extract `extra_params` from the request body on either path. Once populated, the params are merged into the outbound request body by default on the OpenAI-compatible path (Chat Completions and the Responses API), but not on the Anthropic-compatible path
+**Impact**: Provider-specific `extra_params` set on a request may be silently dropped when `use_anthropic_endpoints` is enabled
+**Code**: `BifrostContextKeyPassthroughExtraParams` is set in the OpenAI-compatible branches of `ChatCompletion`/`ChatCompletionStream` and in `Responses`/`ResponsesStream`, but not in the Anthropic-compatible branches, in `core/providers/alibaba/alibaba.go`
+
+
+
+**Severity**: Medium
+**Behavior**: List Models and Embeddings always use Model Studio's OpenAI-compatible endpoints (bearer token auth) regardless of `use_anthropic_endpoints`. Chat Completions and the Responses API use the OpenAI-compatible endpoints by default, or the Anthropic-compatible Messages endpoint (`x-api-key` auth) when `use_anthropic_endpoints` is enabled
+**Impact**: When Anthropic-compatible mode is enabled, parameters and behavior documented for the [OpenAI provider](/providers/supported-providers/openai) no longer apply to Chat Completions or Responses — refer to the [Anthropic provider](/providers/supported-providers/anthropic) docs for those instead
+**Code**: `ChatCompletion`, `ChatCompletionStream`, `Responses`, and `ResponsesStream` in `core/providers/alibaba/alibaba.go` branch on `anthropic.ResolveUseAnthropicEndpoints`; `ListModels` and `Embedding` always delegate to the `openai` package
+
diff --git a/docs/providers/supported-providers/kimi.mdx b/docs/providers/supported-providers/kimi.mdx
new file mode 100644
index 00000000000..7ed1f078f56
--- /dev/null
+++ b/docs/providers/supported-providers/kimi.mdx
@@ -0,0 +1,284 @@
+---
+title: "Kimi (Moonshot AI)"
+description: "Kimi (Moonshot AI) conversion guide - OpenAI-compatible chat by default across the Kimi Open Platform and Kimi Code surfaces, optional per-key/per-alias Anthropic-compatible endpoints, streaming, tool calling, and reasoning"
+icon: "k"
+---
+
+## Overview
+
+Kimi (Moonshot AI) is a provider with a dedicated Bifrost provider implementation. Kimi runs two independent platforms with **separate key systems**: the pay-as-you-go Kimi Open Platform (`api.moonshot.ai`, the default) and the Kimi Code subscription (`api.kimi.com/coding`). By default, Chat Completions use Kimi's **OpenAI-compatible** endpoints. Each key (or an individual alias) can opt into routing Chat Completions and the Responses API through Kimi's **Anthropic-compatible** endpoint instead, using the `use_anthropic_endpoints` toggle. Key characteristics:
+
+- **OpenAI-compatible by default** - Chat Completions use `/chat/completions`, authenticated with a bearer token
+- **Optional Anthropic-compatible mode** - Set `use_anthropic_endpoints` on a key (or override it per-alias) to route Chat Completions and the Responses API through the Anthropic-compatible Messages endpoint, using the shared Anthropic request/response converters
+- **Streaming support** - Server-Sent Events for chat and responses, in both endpoint modes
+- **Tool calling** - Function tools are supported on both the OpenAI-compatible and Anthropic-compatible paths
+- **Reasoning support** - `reasoning_effort` (including `max` on kimi-k3) on the OpenAI-compatible path; Anthropic `thinking` conversion in Anthropic-compatible mode
+
+### Supported Operations
+
+| Operation | Non-Streaming | Streaming | Endpoint (default) | Endpoint (`use_anthropic_endpoints: true`) |
+|-----------|---------------|-----------|---------------------|---------------------------------------------|
+| Chat Completions | ✅ | ✅ | `/chat/completions` | Anthropic Messages endpoint |
+| Responses API | ✅ | ✅ | `/chat/completions` (via Chat Completions fallback) | Anthropic Messages endpoint |
+| Text Completions | ❌ | ❌ | - | - |
+| List Models | ✅ | - | `/models` | `/models` (unaffected) |
+| Embeddings | ❌ | ❌ | - | - |
+| Image Generation | ❌ | ❌ | - | - |
+| Speech (TTS) | ❌ | ❌ | - | - |
+| Transcriptions (STT) | ❌ | ❌ | - | - |
+| Files | ❌ | ❌ | - | - |
+| Batch | ❌ | ❌ | - | - |
+
+
+**Unsupported Operations** (❌): Text Completions, Embeddings, Image Generation, Speech, Transcriptions, Files, Batch, cached content, containers, token counting, compaction, OCR, rerank, video, and passthrough are not supported through this provider. These return `UnsupportedOperationError`. (The Kimi Open Platform does expose Files/Batch APIs upstream, but they are not implemented by this provider.)
+
+
+## Setup & Configuration
+
+Configure Kimi as a provider.
+
+
+
+
+1. Navigate to **Models** > **Model Providers**. Look for **Kimi** under **Configured Providers**. If it is missing, click on **Add New Provider** and select **Kimi**.
+2. Click **Add Key** or edit an existing key.
+3. Set a name for your key.
+4. Paste your API key directly or use an environment variable (for example, `env.KIMI_API_KEY`).
+5. Set **Allowed Models** to **All Models** (default) or the specific model allowlist you want this key to serve.
+6. Leave **Use Anthropic Endpoints** off to use Kimi's OpenAI-compatible endpoints (the default), or turn it on to route Chat Completions and the Responses API through Kimi's Anthropic-compatible endpoint instead. See [Anthropic-Compatible Endpoints](#anthropic-compatible-endpoints-optional) below.
+7. Save the provider configuration.
+
+
+
+
+```json
+{
+ "providers": {
+ "kimi": {
+ "keys": [
+ {
+ "name": "kimi-key-1",
+ "value": "env.KIMI_API_KEY",
+ "models": [
+ "*"
+ ],
+ "weight": 1.0
+ }
+ ]
+ }
+ }
+}
+```
+
+
+
+Refer to the API documentation for [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider).
+
+
+
+```go
+case schemas.Kimi:
+ return []schemas.Key{{
+ Name: "kimi-key-1",
+ Value: *schemas.NewSecretVar("env.KIMI_API_KEY"),
+ Models: []string{"*"},
+ Weight: 1.0,
+ }}, nil
+```
+
+
+
+
+### Base URL (platform selection)
+
+The two Kimi platforms use separate keys that are **not interchangeable** — an Open Platform key never works on Kimi Code and vice versa. The default base URL is the Kimi Open Platform (international). Override `network_config.base_url` to target another surface:
+
+| Surface | `base_url` |
+|---------|------------|
+| Kimi Open Platform, international (default) | `https://api.moonshot.ai/v1` |
+| Kimi Open Platform, China | `https://api.moonshot.cn/v1` |
+| Kimi Code (subscription) | `https://api.kimi.com/coding/v1` |
+
+The Anthropic-compatible base URL is derived automatically from the configured OpenAI-compatible base URL. The derivation is idempotent: a `base_url` already ending in `/anthropic` (the mount itself) is used as-is rather than getting the suffix appended a second time.
+
+Model IDs differ per platform: the Open Platform serves catalog IDs (`kimi-k3`, `kimi-k2.6`, `kimi-k2.7-code`, …), while Kimi Code serves stable aliases (`k3`, `k3-256k`, `kimi-for-coding`, `kimi-for-coding-highspeed`) whose backing model can be upgraded server-side.
+
+
+Kimi Code is a membership benefit intended for interactive coding tools; routing general gateway traffic through it may breach the vendor's terms of service.
+
+
+---
+
+## Anthropic-Compatible Endpoints (optional)
+
+Kimi exposes an Anthropic-compatible Messages endpoint on both platforms. Setting `use_anthropic_endpoints` routes Chat Completions and the Responses API through that endpoint instead — List Models is unaffected and always uses the OpenAI-compatible endpoint.
+
+The setting can be configured per key, and overridden per model alias:
+
+- **Key-level** - Sets the default endpoint mode for every request made with that key.
+- **Alias-level** - Overrides the key-level default for a single alias, so one key can serve some aliases through the OpenAI-compatible endpoints and others through the Anthropic-compatible endpoint.
+
+If neither is set, requests fall back to Kimi's OpenAI-compatible endpoints.
+
+
+Kimi's Anthropic-compatible contract is not officially specified. Empirically it accepts the standard Messages fields and streams the usual SSE events, but **temperature is rescaled upstream (`real = request × 0.6`)**, and the 1M-context variant of the flagship is addressed as `kimi-k3[1m]` on this mount.
+
+
+**Parameter surface on the Anthropic mount** (verified against the vendor docs and live upstream behavior 2026-08-16): the mount documents **no effort equivalent** — K3's `reasoning_effort` is an OpenAI-mount field — so `output_config.effort` is stripped (the mount tolerates unknown fields rather than 400ing, but they are no-ops). `thinking: {"type": "enabled"|"disabled"}` is honored — including `disabled`, which returns a plain non-thinking answer; multi-turn tool loops with thinking on require full `reasoning_content` replay (MoonshotAI/Kimi-K2#129). `metadata.user_id` is tolerated. Anthropic-only extensions (`speed`, `inference_geo`, `service_tier`, `container`, `mcp_servers`, `context_management`, `output_config.task_budget`, `output_config.format`) are stripped.
+
+
+
+
+On the key form, toggle **Use Anthropic Endpoints** (off by default). To override this for a specific alias, open that alias's expanded row in the deployments table and toggle **Use Anthropic endpoints** under **Kimi overrides** — this takes priority over the key-level setting for that alias only.
+
+
+
+The `use_anthropic_endpoints` boolean is part of the same key payload used by [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider), and the alias payload for that key's `models` entries.
+
+
+
+```json
+{
+ "providers": {
+ "kimi": {
+ "keys": [
+ {
+ "name": "kimi-key-1",
+ "value": "env.KIMI_API_KEY",
+ "models": [
+ "*"
+ ],
+ "weight": 1.0,
+ "use_anthropic_endpoints": true
+ }
+ ]
+ }
+ }
+}
+```
+
+To override this per-alias (for example, on a virtual key's model config), set `use_anthropic_endpoints` alongside the alias's `model_id`:
+
+```json
+{
+ "model_id": "kimi-k3",
+ "use_anthropic_endpoints": false
+}
+```
+
+| Field | Type | Required | Description |
+|-------|------|----------|--------------|
+| `use_anthropic_endpoints` | boolean | No | Routes chat completions and responses requests through Anthropic-compatible endpoints. Default: `false`. |
+
+
+
+
+---
+
+# 1. Chat Completions
+
+## Request Parameters
+
+By default, Chat Completions use Kimi's OpenAI-compatible `/chat/completions` endpoint, authenticated with `Authorization: Bearer `. For the full parameter reference and message conversion behavior, see [OpenAI Chat Completions](/providers/supported-providers/openai#1-chat-completions).
+
+When `use_anthropic_endpoints` is enabled, requests are sent instead to Kimi's Anthropic-compatible Messages endpoint (still authenticated with `Authorization: Bearer ` — both Kimi mounts use bearer auth) and built using the shared Anthropic converters. For that parameter reference and message conversion behavior, see [Anthropic Chat Completions](/providers/supported-providers/anthropic#1-chat-completions).
+
+### Reasoning Parameter
+
+- **Default (OpenAI-compatible):** `reasoning_effort` is forwarded for the models that accept it. On kimi-k3 the full scale is supported, including `max` (the model's native default); on other models the field is stripped rather than rejected by the vendor. K2.x thinking is controlled via the `thinking` extra param instead.
+- **`use_anthropic_endpoints: true`:** Reasoning/thinking parameters are mapped through the Anthropic converters (`reasoning` → `thinking`), the same as the [Anthropic provider](/providers/supported-providers/anthropic#1-chat-completions).
+
+### Extra Parameters
+
+Kimi enables passthrough extra parameters for Chat Completions on the default OpenAI-compatible endpoints. Kimi-specific fields such as `thinking` (`{"type": "enabled", "keep": "all"}` on K2.x), `prompt_cache_key`, `safety_identifier`, or message-level `partial` can be sent under `extra_params` with the `x-bf-passthrough-extra-params: true` header. Extra parameters are **not** passed through by default when `use_anthropic_endpoints` is enabled.
+
+
+The Kimi Code endpoint requires `prompt_cache_key` for context-cache hits — without it, every request pays full input price. Map a stable session/task identifier to it via `extra_params`.
+
+
+---
+
+# 2. Responses API
+
+- **Default (OpenAI-compatible):** Responses requests fall back to Chat Completions, the same conversion pattern used by other OpenAI-compatible-only providers (Kimi exposes no `/responses` endpoint):
+
+ ```
+ ResponsesRequest → ChatRequest → Response conversion
+ ```
+
+- **`use_anthropic_endpoints: true`:** Responses requests are sent natively to Kimi's Anthropic-compatible Messages endpoint — both non-streaming and streaming requests build an Anthropic-format request body directly and convert the response back to Bifrost's Responses format.
+
+---
+
+# 3. List Models
+
+Lists available models from Kimi through the OpenAI-compatible `/models` endpoint. On the Kimi Code surface the listing reflects the stable alias IDs and should be treated as best-effort; prefer wildcard (`*`) model allowlists on Kimi Code keys.
+
+---
+
+## Unsupported Features
+
+| Feature | Reason |
+|---------|--------|
+| Text Completions | Not offered by Kimi through this provider |
+| Embeddings | Not offered by Kimi |
+| Image Generation | Not offered by Kimi through this provider |
+| Speech/TTS | Not offered by Kimi through this provider |
+| Transcription/STT | Not offered by Kimi through this provider |
+| Batch Operations | Open Platform exposes batches upstream; not implemented by this provider |
+| File Management | Open Platform exposes files upstream; not implemented by this provider |
+| Cached Content | Only Gemini and Vertex AI support cached content in Bifrost |
+| Container Management | Not offered by Kimi through this provider |
+| Token Counting | Kimi's token-counting endpoint uses a Kimi-specific contract not covered by Bifrost's count-tokens handlers |
+| Rerank/OCR/Video | Not offered by Kimi through this provider |
+
+---
+
+## Caveats
+
+
+**Severity**: Low
+**Behavior**: Kimi defaults to `https://api.moonshot.ai/v1` (Open Platform, international)
+**Impact**: Kimi Code subscription keys and China-region keys must override `network_config.base_url`; keys are platform-specific and a key/host mismatch returns 401
+**Code**: `NewKimiProvider` sets the default base URL when no provider-level base URL is configured
+
+
+
+**Severity**: Low
+**Behavior**: The Anthropic-compatible base URL is derived from the configured OpenAI-compatible base URL (Open Platform hosts and the Kimi Code host are both recognized)
+**Impact**: A completely custom OpenAI-compatible host that matches none of the known shapes may not derive a valid Anthropic mount URL
+**Code**: `deriveAnthropicBaseURL` in `core/providers/kimi/utils.go`
+
+
+
+**Severity**: Low
+**Behavior**: An alias-level `use_anthropic_endpoints` override always wins over the key-level setting for that alias; if neither is set, requests default to the OpenAI-compatible endpoints
+**Impact**: A single key can serve some aliases through OpenAI-compatible endpoints and others through the Anthropic-compatible endpoint
+**Code**: `anthropic.ResolveUseAnthropicEndpoints` in `core/providers/anthropic/utils.go`, used by `ChatCompletion`, `ChatCompletionStream`, `Responses`, and `ResponsesStream` in `core/providers/kimi/kimi.go`
+
+
+
+**Severity**: Medium
+**Behavior**: Kimi's Anthropic-compatible endpoint multiplies the requested temperature by 0.6 upstream (unofficial, undocumented contract)
+**Impact**: Sampling behavior differs between the two endpoint modes for the same request
+
+
+
+**Severity**: Medium
+**Behavior**: Kimi K-series models fix `temperature`/`top_p`/`n`/penalty parameters per model; explicitly sending non-default values errors upstream. Bifrost never injects defaults for these fields
+**Impact**: SDK clients that auto-send `temperature: 1` (or similar) will fail against K-series models; omit sampling params for these models
+
+
+
+**Severity**: Low
+**Behavior**: Extra parameters are merged into the outbound request body by default on the OpenAI-compatible path (Chat Completions), but not on the Anthropic-compatible path
+**Impact**: Provider-specific `extra_params` set on a request may be silently dropped when `use_anthropic_endpoints` is enabled
+**Code**: `BifrostContextKeyPassthroughExtraParams` is set in the OpenAI-compatible branches of `ChatCompletion`/`ChatCompletionStream`, but not in the Anthropic-compatible branches, in `core/providers/kimi/kimi.go`
+
+
+
+**Severity**: Medium
+**Behavior**: List Models always uses Kimi's OpenAI-compatible endpoint regardless of `use_anthropic_endpoints`. Chat Completions and the Responses API use the OpenAI-compatible endpoints by default, or the Anthropic-compatible Messages endpoint when `use_anthropic_endpoints` is enabled. Both mounts authenticate with `Authorization: Bearer `
+**Impact**: When Anthropic-compatible mode is enabled, parameters and behavior documented for the [OpenAI provider](/providers/supported-providers/openai) no longer apply to Chat Completions or Responses — refer to the [Anthropic provider](/providers/supported-providers/anthropic) docs for those instead
+**Code**: `ChatCompletion`, `ChatCompletionStream`, `Responses`, and `ResponsesStream` in `core/providers/kimi/kimi.go` branch on `anthropic.ResolveUseAnthropicEndpoints`; `ListModels` always delegates to the `openai` package
+
diff --git a/docs/providers/supported-providers/overview.mdx b/docs/providers/supported-providers/overview.mdx
index 2943137dc41..d57365d0c23 100644
--- a/docs/providers/supported-providers/overview.mdx
+++ b/docs/providers/supported-providers/overview.mdx
@@ -16,6 +16,7 @@ The following table summarizes which operations are supported by each provider v
| Provider | Models | Text | Text (stream) | Chat | Chat (stream) | Responses | Responses (stream) | Images | Images (stream) | Image Edit | Image Edit (stream) | Image Variation | Embeddings | TTS | TTS (stream) | STT | STT (stream) | Files | Batch | Count tokens | Rerank | OCR | Video | Video Remix | Containers | Passthrough | Passthrough (stream) |
| ------------------------------------ | ------ | ---- | ------------- | ---- | ------------- | --------- | ------------------ | ------ | --------------- | ---------- | ------------------- | --------------- | ---------- | --- | ------------ | --- | ------------ | ----- | ----- | ------------ | ------ | --- | ----- | ----------- | ---------- | ----------- | -------------------- |
+| Alibaba (`alibaba/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
| Anthropic (`anthropic/`) | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ |
| Azure (`azure/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ |
| Bedrock (`bedrock/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
@@ -28,6 +29,7 @@ The following table summarizes which operations are supported by each provider v
| Gemini (`gemini/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ |
| Groq (`groq/`) | ✅ | 🟡 | 🟡 | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
| Hugging Face (`huggingface/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
+| Kimi (`kimi/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
| Mistral (`mistral/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ |
| Nebius (`nebius/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
| Ollama (`ollama/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
@@ -44,6 +46,7 @@ The following table summarizes which operations are supported by each provider v
| vLLM (`vllm/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
| Wafer (`wafer/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
| xAI (`xai/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
+| Zhipu (`zhipu/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
- 🟡 Not supported by the downstream provider, but internally implemented by Bifrost as a fallback.
- ❌ Not supported by the downstream provider, hence not supported by Bifrost.
diff --git a/docs/providers/supported-providers/zhipu.mdx b/docs/providers/supported-providers/zhipu.mdx
new file mode 100644
index 00000000000..26551145fbe
--- /dev/null
+++ b/docs/providers/supported-providers/zhipu.mdx
@@ -0,0 +1,287 @@
+---
+title: "Zhipu AI (Z.AI / GLM)"
+description: "Zhipu AI (Z.AI / GLM) conversion guide - OpenAI-compatible chat by default on the General API and GLM Coding Plan surfaces, optional per-key/per-alias Anthropic-compatible endpoints (Coding Plan), streaming, tool calling, and reasoning"
+icon: "z"
+---
+
+## Overview
+
+Zhipu AI (Z.AI / GLM) is a provider with a dedicated Bifrost provider implementation. Zhipu runs the GLM model family on two storefronts with separate key pools — **Z.AI** (international, `api.z.ai`, the default) and **BigModel** (China, `open.bigmodel.cn`) — each with a pay-as-you-go General API surface and a GLM Coding Plan subscription surface. By default, Chat Completions use Zhipu's **OpenAI-compatible** endpoints. Each key (or an individual alias) can opt into routing Chat Completions and the Responses API through the **Anthropic-compatible** endpoint instead, using the `use_anthropic_endpoints` toggle. Key characteristics:
+
+- **OpenAI-compatible by default** - Chat Completions use `/chat/completions`, authenticated with a bearer token
+- **Optional Anthropic-compatible mode** - Set `use_anthropic_endpoints` on a key (or override it per-alias) to route Chat Completions and the Responses API through the Anthropic-compatible Messages endpoint. The Anthropic mount is only offered on the **GLM Coding Plan** surface — point `base_url` at a Coding Plan host before enabling it
+- **Streaming support** - Server-Sent Events for chat and responses, in both endpoint modes
+- **Tool calling** - Function tools are supported on both the OpenAI-compatible and Anthropic-compatible paths
+- **Reasoning support** - GLM models think by default; `reasoning_effort` is honored on GLM-5.2+ and stripped elsewhere. GLM-5.3 narrowed the enum to `max`/`high`/`low` — wider tiers are mapped for you (`xhigh`→`max`, `medium`→`high`, `minimal`/`none`→`low`). Preserved-thinking is controlled via the `thinking` extra param (`{"type": "enabled", "clear_thinking": false}`); GLM-5.3 no longer accepts `thinking.type: "disabled"` (thinking is always on)
+
+### Supported Operations
+
+| Operation | Non-Streaming | Streaming | Endpoint (default) | Endpoint (`use_anthropic_endpoints: true`) |
+|-----------|---------------|-----------|---------------------|---------------------------------------------|
+| Chat Completions | ✅ | ✅ | `/chat/completions` | Anthropic Messages endpoint (Coding Plan) |
+| Responses API | ✅ | ✅ | `/chat/completions` (via Chat Completions fallback) | Anthropic Messages endpoint (Coding Plan) |
+| Text Completions | ❌ | ❌ | - | - |
+| List Models | ✅ | - | `/models` | `/models` (unaffected) |
+| Embeddings | ❌ | ❌ | - | - |
+| Image Generation | ❌ | ❌ | - | - |
+| Speech (TTS) | ❌ | ❌ | - | - |
+| Transcriptions (STT) | ❌ | ❌ | - | - |
+| Files | ❌ | ❌ | - | - |
+| Batch | ❌ | ❌ | - | - |
+
+
+**Unsupported Operations** (❌): Text Completions, Embeddings, Image Generation, Speech, Transcriptions, Files, Batch, cached content, containers, token counting, compaction, OCR, rerank, video, and passthrough are not supported through this provider. These return `UnsupportedOperationError`. (Embeddings exist on the China BigModel platform only — not on the Z.AI international surface this provider defaults to.)
+
+
+## Setup & Configuration
+
+Configure Zhipu as a provider.
+
+
+
+
+1. Navigate to **Models** > **Model Providers**. Look for **Zhipu** under **Configured Providers**. If it is missing, click on **Add New Provider** and select **Zhipu**.
+2. Click **Add Key** or edit an existing key.
+3. Set a name for your key.
+4. Paste your API key directly or use an environment variable (for example, `env.ZHIPU_API_KEY`).
+5. Set **Allowed Models** to **All Models** (default) or the specific model allowlist you want this key to serve.
+6. Leave **Use Anthropic Endpoints** off to use Zhipu's OpenAI-compatible endpoints (the default), or turn it on to route Chat Completions and the Responses API through the Anthropic-compatible endpoint instead (GLM Coding Plan keys only). See [Anthropic-Compatible Endpoints](#anthropic-compatible-endpoints-optional) below.
+7. Save the provider configuration.
+
+
+
+
+```json
+{
+ "providers": {
+ "zhipu": {
+ "keys": [
+ {
+ "name": "zhipu-key-1",
+ "value": "env.ZHIPU_API_KEY",
+ "models": [
+ "*"
+ ],
+ "weight": 1.0
+ }
+ ]
+ }
+ }
+}
+```
+
+
+
+Refer to the API documentation for [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider).
+
+
+
+```go
+case schemas.Zhipu:
+ return []schemas.Key{{
+ Name: "zhipu-key-1",
+ Value: *schemas.NewSecretVar("env.ZHIPU_API_KEY"),
+ Models: []string{"*"},
+ Weight: 1.0,
+ }}, nil
+```
+
+
+
+
+### Base URL (storefront & plan)
+
+General API keys and GLM Coding Plan keys are **not interchangeable**. The default base URL is the Z.AI international General API. Override `network_config.base_url` to target another surface:
+
+| Surface | `base_url` |
+|---------|------------|
+| General API, international (default) | `https://api.z.ai/api/paas/v4` |
+| General API, China | `https://open.bigmodel.cn/api/paas/v4` |
+| GLM Coding Plan, international | `https://api.z.ai/api/coding/paas/v4` |
+| GLM Coding Plan, China | `https://open.bigmodel.cn/api/coding/paas/v4` |
+
+The Anthropic-compatible base URL is derived automatically from the configured OpenAI-compatible base URL (Coding Plan hosts derive their Anthropic mount; the General API has no Anthropic mount). The derivation is idempotent: a `base_url` already ending in `/anthropic` (the mount itself) is used as-is rather than getting the suffix appended a second time.
+
+
+The GLM Coding Plan is a subscription intended for interactive coding tools; routing general gateway traffic through it may breach the vendor's terms of service. When a plan lapses, Coding-Plan-only models (e.g. glm-5.2 on the Anthropic mount) fail with a clean vendor 429 until the subscription is renewed.
+
+
+---
+
+## Anthropic-Compatible Endpoints (optional)
+
+Zhipu exposes an Anthropic-compatible Messages endpoint on the **GLM Coding Plan** surface only. Setting `use_anthropic_endpoints` routes Chat Completions and the Responses API through that endpoint instead — List Models is unaffected and always uses the OpenAI-compatible endpoint.
+
+The setting can be configured per key, and overridden per model alias:
+
+- **Key-level** - Sets the default endpoint mode for every request made with that key.
+- **Alias-level** - Overrides the key-level default for a single alias, so one key can serve some aliases through the OpenAI-compatible endpoints and others through the Anthropic-compatible endpoint.
+
+If neither is set, requests fall back to Zhipu's OpenAI-compatible endpoints.
+
+**Parameter surface on the Anthropic mount** (verified against the vendor docs and live upstream behavior 2026-08-16):
+
+- `output_config.effort` is forwarded for **GLM-5.2 and later** — z.ai accepts the field on this mount and maps out-of-scale values server-side (GLM-5.3: `none`/`minimal`/`low`→`low`, `medium`/`high`→`high`, `xhigh`/`max`→`max`), so any Bifrost effort value is passed through verbatim. For older models (glm-5/5.1/4.7/4.6) the field is stripped.
+- **The mount treats an absent `thinking` field as disabled.** GLM-5.3+ rejects that with a 1210 error ("This model always engages in thinking and cannot be disabled"), so for GLM-5.3+ a `thinking: {"type": "enabled"}` is always sent — synthesized when missing (budget derived from `output_config.effort`, else the minimum), and `thinking.type: "disabled"` is rewritten to `enabled` with the minimum budget (the model thinks regardless — there is no off mode). GLM-4.7 is forced-thinking too (`disabled` is rewritten), but tolerates an absent field.
+- `thinking: {"type": "enabled", "budget_tokens": N}`, `metadata.user_id`, `system`/message `cache_control`, `tools`, `tool_choice`, `stop_sequences`, `top_p`, and `top_k` pass through.
+- Anthropic-only extensions (`speed`, `inference_geo`, `service_tier`, `container`, `mcp_servers`, `context_management`, `output_config.task_budget`) are stripped — the mount does not accept them.
+
+
+
+
+On the key form, toggle **Use Anthropic Endpoints** (off by default). To override this for a specific alias, open that alias's expanded row in the deployments table and toggle **Use Anthropic endpoints** under **Zhipu overrides** — this takes priority over the key-level setting for that alias only.
+
+
+
+The `use_anthropic_endpoints` boolean is part of the same key payload used by [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider), and the alias payload for that key's `models` entries.
+
+
+
+```json
+{
+ "providers": {
+ "zhipu": {
+ "keys": [
+ {
+ "name": "zhipu-key-1",
+ "value": "env.ZHIPU_API_KEY",
+ "models": [
+ "*"
+ ],
+ "weight": 1.0,
+ "use_anthropic_endpoints": true
+ }
+ ],
+ "network_config": {
+ "base_url": "https://api.z.ai/api/coding/paas/v4"
+ }
+ }
+ }
+}
+```
+
+To override this per-alias (for example, on a virtual key's model config), set `use_anthropic_endpoints` alongside the alias's `model_id`:
+
+```json
+{
+ "model_id": "glm-5.2",
+ "use_anthropic_endpoints": false
+}
+```
+
+| Field | Type | Required | Description |
+|-------|------|----------|--------------|
+| `use_anthropic_endpoints` | boolean | No | Routes chat completions and responses requests through Anthropic-compatible endpoints (GLM Coding Plan surface only). Default: `false`. |
+
+
+
+
+---
+
+# 1. Chat Completions
+
+## Request Parameters
+
+By default, Chat Completions use Zhipu's OpenAI-compatible `/chat/completions` endpoint, authenticated with `Authorization: Bearer `. For the full parameter reference and message conversion behavior, see [OpenAI Chat Completions](/providers/supported-providers/openai#1-chat-completions).
+
+When `use_anthropic_endpoints` is enabled, requests are sent instead to the Anthropic-compatible Messages endpoint (still authenticated with `Authorization: Bearer ` — Claude Code connects with an `ANTHROPIC_AUTH_TOKEN`, i.e. bearer auth) and built using the shared Anthropic converters. For that parameter reference and message conversion behavior, see [Anthropic Chat Completions](/providers/supported-providers/anthropic#1-chat-completions).
+
+### Reasoning Parameter
+
+- **Default (OpenAI-compatible):** `reasoning_effort` is forwarded only for GLM-5.2+ (including the Coding Plan 1M-context alias `glm-5.2[1m]`) and stripped for every other model — the vendor rejects the field there. Per-model enums differ:
+ - **GLM-5.2** accepts the full legacy enum (`max`, `xhigh`, `high`, `medium`, `low`, `minimal`, `none`) and maps the extra tiers itself (`low`/`medium`→`high`, `xhigh`→`max`, `none`/`minimal`→thinking off).
+ - **GLM-5.3+** accepts exactly `max` (the default and recommended tier), `high`, and `low` — the API errors on any other value. Bifrost maps wider tiers onto the nearest supported one before forwarding: `xhigh`→`max`, `medium`→`high`, `minimal`/`none`→`low`. (The Coding Plan performs the same mapping server-side.)
+ - **GLM-5.3 forced thinking:** `thinking.type: "disabled"` is rejected by the vendor for GLM-5.3 — thinking cannot be turned off on that model. Sending `thinking: {"type": "enabled"}` (or omitting `thinking`) is fine.
+- **`use_anthropic_endpoints: true`:** Reasoning/thinking parameters are mapped through the Anthropic converters (`reasoning` → `thinking`), the same as the [Anthropic provider](/providers/supported-providers/anthropic#1-chat-completions).
+
+### Extra Parameters
+
+Zhipu enables passthrough extra parameters for Chat Completions on the default OpenAI-compatible endpoints. Zhipu-specific fields such as `thinking` (`{"type": "enabled", "clear_thinking": false}` for preserved-thinking multi-turn replay), `do_sample`, or `request_id` can be sent under `extra_params` with the `x-bf-passthrough-extra-params: true` header. Extra parameters are **not** passed through by default when `use_anthropic_endpoints` is enabled.
+
+---
+
+# 2. Responses API
+
+- **Default (OpenAI-compatible):** Responses requests fall back to Chat Completions, the same conversion pattern used by other OpenAI-compatible-only providers (Zhipu exposes no `/responses` endpoint):
+
+ ```
+ ResponsesRequest → ChatRequest → Response conversion
+ ```
+
+- **`use_anthropic_endpoints: true`:** Responses requests are sent natively to the Anthropic-compatible Messages endpoint — both non-streaming and streaming requests build an Anthropic-format request body directly and convert the response back to Bifrost's Responses format.
+
+---
+
+# 3. List Models
+
+Lists available models from Zhipu through the OpenAI-compatible `/models` endpoint. The Anthropic-compatible mount has no models endpoint, so List Models always uses the OpenAI-compatible mount regardless of `use_anthropic_endpoints`.
+
+---
+
+## Unsupported Features
+
+| Feature | Reason |
+|---------|--------|
+| Text Completions | Not offered by Zhipu through this provider |
+| Embeddings | Offered on the China BigModel platform only, not on Z.AI international |
+| Image Generation | Not offered through this provider |
+| Speech/TTS | Not offered through this provider |
+| Transcription/STT | Not offered through this provider |
+| Batch Operations | Not offered through this provider |
+| File Management | Not offered through this provider |
+| Cached Content | Only Gemini and Vertex AI support cached content in Bifrost |
+| Container Management | Not offered by Zhipu through this provider |
+| Token Counting | Not offered by Zhipu through this provider |
+| Rerank/OCR/Video | Not offered through this provider |
+
+---
+
+## Caveats
+
+
+**Severity**: Low
+**Behavior**: Zhipu defaults to `https://api.z.ai/api/paas/v4` (Z.AI international General API)
+**Impact**: BigModel (China) keys and GLM Coding Plan keys must override `network_config.base_url`; a key/host mismatch returns 401
+**Code**: `NewZhipuProvider` sets the default base URL when no provider-level base URL is configured
+
+
+
+**Severity**: Medium
+**Behavior**: The Anthropic-compatible Messages endpoint exists only on the GLM Coding Plan surface. Enabling `use_anthropic_endpoints` with a General API `base_url` derives no valid mount
+**Impact**: Point `base_url` at a Coding Plan host (e.g. `https://api.z.ai/api/coding/paas/v4`) before enabling the toggle
+**Code**: `deriveAnthropicBaseURL` in `core/providers/zhipu/utils.go`
+
+
+
+**Severity**: Low
+**Behavior**: An alias-level `use_anthropic_endpoints` override always wins over the key-level setting for that alias; if neither is set, requests default to the OpenAI-compatible endpoints
+**Impact**: A single key can serve some aliases through OpenAI-compatible endpoints and others through the Anthropic-compatible endpoint
+**Code**: `anthropic.ResolveUseAnthropicEndpoints` in `core/providers/anthropic/utils.go`, used by `ChatCompletion`, `ChatCompletionStream`, `Responses`, and `ResponsesStream` in `core/providers/zhipu/zhipu.go`
+
+
+
+**Severity**: Medium
+**Behavior**: On the OpenAI-compatible path, `reasoning_effort` is only forwarded for GLM-5.2-and-later models; it is stripped for all others (the vendor rejects the field there). For GLM-5.3-and-later, values outside the vendor's `max`/`high`/`low` enum are mapped onto the nearest tier before forwarding
+**Impact**: A `reasoning_effort` set on a GLM-4.x model is silently dropped (those models always think); a legacy tier sent to GLM-5.3+ is coerced (e.g. `medium`→`high`) instead of surfacing a vendor 400
+**Code**: `applyZhipuReasoning` in `core/providers/openai/chat.go`
+
+
+
+**Severity**: Low
+**Behavior**: GLM models stream reasoning token-by-token, so a single response can produce hundreds of SSE chunks, and tool calls arrive only after the full reasoning trace
+**Impact**: Clients that cap or batch stream chunks should use generous limits; this is normal vendor behavior, not a stalled stream
+
+
+
+**Severity**: Low
+**Behavior**: Extra parameters are merged into the outbound request body by default on the OpenAI-compatible path (Chat Completions), but not on the Anthropic-compatible path
+**Impact**: Provider-specific `extra_params` set on a request may be silently dropped when `use_anthropic_endpoints` is enabled
+**Code**: `BifrostContextKeyPassthroughExtraParams` is set in the OpenAI-compatible branches of `ChatCompletion`/`ChatCompletionStream`, but not in the Anthropic-compatible branches, in `core/providers/zhipu/zhipu.go`
+
+
+
+**Severity**: Medium
+**Behavior**: List Models always uses Zhipu's OpenAI-compatible endpoint regardless of `use_anthropic_endpoints`. Chat Completions and the Responses API use the OpenAI-compatible endpoints by default, or the Anthropic-compatible Messages endpoint (Coding Plan) when `use_anthropic_endpoints` is enabled. Both mounts authenticate with `Authorization: Bearer `
+**Impact**: When Anthropic-compatible mode is enabled, parameters and behavior documented for the [OpenAI provider](/providers/supported-providers/openai) no longer apply to Chat Completions or Responses — refer to the [Anthropic provider](/providers/supported-providers/anthropic) docs for those instead
+**Code**: `ChatCompletion`, `ChatCompletionStream`, `Responses`, and `ResponsesStream` in `core/providers/zhipu/zhipu.go` branch on `anthropic.ResolveUseAnthropicEndpoints`; `ListModels` always delegates to the `openai` package
+
diff --git a/tests/e2e/api/collections/provider-harness.json b/tests/e2e/api/collections/provider-harness.json
index 7a4e58fa0b3..17d1012bac8 100644
--- a/tests/e2e/api/collections/provider-harness.json
+++ b/tests/e2e/api/collections/provider-harness.json
@@ -136486,6 +136486,112 @@
}
]
},
+ {
+ "name": "58. Alibaba qwen3.8-max reasoning_effort xhigh top tier (PR #6054)",
+ "description": "PR #6054: qwen3.8-max on the Model Studio OpenAI-compatible mount natively accepts reasoning_effort none/minimal/low/medium/high/xhigh and 400'd \"max\" until ~2026-08-22 (vendor error: 'reasoning_effort' must be one of...). The gateway originally published the ladder [low, medium, high, max], so both a requested \"xhigh\" and \"max\" hit the wire as \"max\". The ladder now tops out at xhigh: xhigh forwards verbatim and max clamps down to xhigh via the shared effort normalizer (chat + responses paths). Chat completions do not reliably echo the effort, so the cases pin the absence of the vendor enum 400 and, when the vendor does echo, that it shows xhigh.",
+ "item": [
+ {
+ "name": "alibaba/qwen3.8-max /v1/chat/completions reasoning_effort=xhigh forwarded as the top tier (2xx, no enum 400) - PR #6054",
+ "request": {
+ "method": "POST",
+ "header": [
+ {
+ "key": "Content-Type",
+ "value": "application/json"
+ }
+ ],
+ "body": {
+ "mode": "raw",
+ "raw": "{\n \"model\": \"alibaba/qwen3.8-max\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with the single word: ok\"\n }\n ],\n \"reasoning_effort\": \"xhigh\",\n \"max_tokens\": 64\n}"
+ },
+ "url": {
+ "raw": "{{baseUrl}}/v1/chat/completions",
+ "host": [
+ "{{baseUrl}}"
+ ],
+ "path": [
+ "v1",
+ "chat",
+ "completions"
+ ]
+ }
+ },
+ "event": [
+ {
+ "listen": "test",
+ "script": {
+ "type": "text/javascript",
+ "exec": [
+ "if ([401, 403, 429, 500, 502, 503, 504, 529].indexOf(pm.response.code) !== -1) { return; }",
+ "pm.test('qwen3.8-max xhigh: 2xx (a 400 means the effort value was mangled before the wire)', function () {",
+ " pm.expect(pm.response.code, 'failed: ' + pm.response.text()).to.be.below(400);",
+ "});",
+ "pm.test('qwen3.8-max xhigh: no vendor reasoning_effort enum 400', function () {",
+ " pm.expect(pm.response.text().indexOf(\"'reasoning_effort' must be one of\"), 'vendor rejected the effort: ' + pm.response.text()).to.eql(-1);",
+ "});",
+ "if (pm.response.code >= 400) { return; }",
+ "var r = pm.response.json();",
+ "pm.test('qwen3.8-max xhigh: echo consistent with xhigh when present', function () {",
+ " if (r.reasoning_effort) { pm.expect(r.reasoning_effort, 'echo: ' + pm.response.text()).to.eql('xhigh'); }",
+ " if (r.reasoning && r.reasoning.effort) { pm.expect(r.reasoning.effort, 'echo: ' + pm.response.text()).to.eql('xhigh'); }",
+ "});"
+ ]
+ }
+ }
+ ]
+ },
+ {
+ "name": "alibaba/qwen3.8-max /v1/chat/completions reasoning_effort=max clamped to xhigh on the wire (2xx) - PR #6054",
+ "request": {
+ "method": "POST",
+ "header": [
+ {
+ "key": "Content-Type",
+ "value": "application/json"
+ }
+ ],
+ "body": {
+ "mode": "raw",
+ "raw": "{\n \"model\": \"alibaba/qwen3.8-max\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with the single word: ok\"\n }\n ],\n \"reasoning_effort\": \"max\",\n \"max_tokens\": 64\n}"
+ },
+ "url": {
+ "raw": "{{baseUrl}}/v1/chat/completions",
+ "host": [
+ "{{baseUrl}}"
+ ],
+ "path": [
+ "v1",
+ "chat",
+ "completions"
+ ]
+ }
+ },
+ "event": [
+ {
+ "listen": "test",
+ "script": {
+ "type": "text/javascript",
+ "exec": [
+ "if ([401, 403, 429, 500, 502, 503, 504, 529].indexOf(pm.response.code) !== -1) { return; }",
+ "pm.test('qwen3.8-max max: 2xx (a 400 means the effort value was mangled before the wire)', function () {",
+ " pm.expect(pm.response.code, 'failed: ' + pm.response.text()).to.be.below(400);",
+ "});",
+ "pm.test('qwen3.8-max max: no vendor reasoning_effort enum 400', function () {",
+ " pm.expect(pm.response.text().indexOf(\"'reasoning_effort' must be one of\"), 'vendor rejected the effort: ' + pm.response.text()).to.eql(-1);",
+ "});",
+ "if (pm.response.code >= 400) { return; }",
+ "var r = pm.response.json();",
+ "pm.test('qwen3.8-max max: echo consistent with xhigh when present', function () {",
+ " if (r.reasoning_effort) { pm.expect(r.reasoning_effort, 'echo: ' + pm.response.text()).to.eql('xhigh'); }",
+ " if (r.reasoning && r.reasoning.effort) { pm.expect(r.reasoning.effort, 'echo: ' + pm.response.text()).to.eql('xhigh'); }",
+ "});"
+ ]
+ }
+ }
+ ]
+ }
+ ]
+ },
{
"name": "58. Typed embedding envelopes and SDK alias (PR #6335)",
"item": [
diff --git a/transports/config.schema.json b/transports/config.schema.json
index aceb8b4fc99..1aeb0ebdbec 100644
--- a/transports/config.schema.json
+++ b/transports/config.schema.json
@@ -465,6 +465,15 @@
},
"sarvam": {
"$ref": "#/$defs/provider"
+ },
+ "alibaba": {
+ "$ref": "#/$defs/provider_with_alibaba_config"
+ },
+ "kimi": {
+ "$ref": "#/$defs/provider_with_kimi_config"
+ },
+ "zhipu": {
+ "$ref": "#/$defs/provider_with_zhipu_config"
}
},
"additionalProperties": true
@@ -2554,7 +2563,8 @@
"perplexity",
"replicate",
"sgl",
- "huggingface"
+ "huggingface",
+ "alibaba"
]
},
"embedding_model": {
@@ -4549,6 +4559,57 @@
}
]
},
+ "alibaba_key": {
+ "allOf": [
+ {
+ "$ref": "#/$defs/base_key"
+ },
+ {
+ "type": "object",
+ "properties": {
+ "use_anthropic_endpoints": {
+ "type": "boolean",
+ "description": "Routes chat completions and responses requests through Anthropic-compatible endpoints.",
+ "default": false
+ }
+ }
+ }
+ ]
+ },
+ "kimi_key": {
+ "allOf": [
+ {
+ "$ref": "#/$defs/base_key"
+ },
+ {
+ "type": "object",
+ "properties": {
+ "use_anthropic_endpoints": {
+ "type": "boolean",
+ "description": "Routes chat completions and Responses requests through Anthropic-compatible endpoints (Open Platform /v1 is mapped to the /anthropic mount; Kimi Code shares the coding base URL).",
+ "default": false
+ }
+ }
+ }
+ ]
+ },
+ "zhipu_key": {
+ "allOf": [
+ {
+ "$ref": "#/$defs/base_key"
+ },
+ {
+ "type": "object",
+ "properties": {
+ "use_anthropic_endpoints": {
+ "type": "boolean",
+ "description": "Routes chat completions and Responses requests through Anthropic-compatible endpoints. Requires a GLM Coding Plan base URL (…/api/coding/paas/v4) — the Anthropic-compatible endpoint rejects General API keys.",
+ "default": false
+ }
+ }
+ }
+ ]
+ },
"fireworks_key": {
"allOf": [
{
@@ -5048,6 +5109,123 @@
"required": ["keys"],
"additionalProperties": false
},
+ "provider_with_alibaba_config": {
+ "type": "object",
+ "properties": {
+ "keys": {
+ "type": "array",
+ "items": {
+ "$ref": "#/$defs/alibaba_key"
+ },
+ "minItems": 1,
+ "description": "API keys for this provider"
+ },
+ "network_config": {
+ "$ref": "#/$defs/network_config"
+ },
+ "concurrency_and_buffer_size": {
+ "$ref": "#/$defs/concurrency_and_buffer_size"
+ },
+ "proxy_config": {
+ "$ref": "#/$defs/proxy_config"
+ },
+ "send_back_raw_request": {
+ "type": "boolean",
+ "description": "Include raw request in BifrostResponse (default: false)"
+ },
+ "send_back_raw_response": {
+ "type": "boolean",
+ "description": "Include raw response in BifrostResponse (default: false)"
+ },
+ "store_raw_request_response": {
+ "type": "boolean",
+ "description": "Capture raw request/response for internal logging only; strip from API responses returned to clients (default: false)"
+ },
+ "custom_provider_config": {
+ "$ref": "#/$defs/custom_provider_config"
+ }
+ },
+ "required": ["keys"],
+ "additionalProperties": false
+ },
+ "provider_with_kimi_config": {
+ "type": "object",
+ "properties": {
+ "keys": {
+ "type": "array",
+ "items": {
+ "$ref": "#/$defs/kimi_key"
+ },
+ "minItems": 1,
+ "description": "API keys for this provider"
+ },
+ "network_config": {
+ "$ref": "#/$defs/network_config"
+ },
+ "concurrency_and_buffer_size": {
+ "$ref": "#/$defs/concurrency_and_buffer_size"
+ },
+ "proxy_config": {
+ "$ref": "#/$defs/proxy_config"
+ },
+ "send_back_raw_request": {
+ "type": "boolean",
+ "description": "Include raw request in BifrostResponse (default: false)"
+ },
+ "send_back_raw_response": {
+ "type": "boolean",
+ "description": "Include raw response in BifrostResponse (default: false)"
+ },
+ "store_raw_request_response": {
+ "type": "boolean",
+ "description": "Capture raw request/response for internal logging only; strip from API responses returned to clients (default: false)"
+ },
+ "custom_provider_config": {
+ "$ref": "#/$defs/custom_provider_config"
+ }
+ },
+ "required": ["keys"],
+ "additionalProperties": false
+ },
+ "provider_with_zhipu_config": {
+ "type": "object",
+ "properties": {
+ "keys": {
+ "type": "array",
+ "items": {
+ "$ref": "#/$defs/zhipu_key"
+ },
+ "minItems": 1,
+ "description": "API keys for this provider"
+ },
+ "network_config": {
+ "$ref": "#/$defs/network_config"
+ },
+ "concurrency_and_buffer_size": {
+ "$ref": "#/$defs/concurrency_and_buffer_size"
+ },
+ "proxy_config": {
+ "$ref": "#/$defs/proxy_config"
+ },
+ "send_back_raw_request": {
+ "type": "boolean",
+ "description": "Include raw request in BifrostResponse (default: false)"
+ },
+ "send_back_raw_response": {
+ "type": "boolean",
+ "description": "Include raw response in BifrostResponse (default: false)"
+ },
+ "store_raw_request_response": {
+ "type": "boolean",
+ "description": "Capture raw request/response for internal logging only; strip from API responses returned to clients (default: false)"
+ },
+ "custom_provider_config": {
+ "$ref": "#/$defs/custom_provider_config"
+ }
+ },
+ "required": ["keys"],
+ "additionalProperties": false
+ },
"provider_with_fireworks_config": {
"type": "object",
"properties": {
@@ -7619,7 +7797,10 @@
"runware",
"fireworks",
"sarvam",
- "wafer"
+ "wafer",
+ "alibaba",
+ "kimi",
+ "zhipu"
],
"description": "Base provider type to extend"
},
diff --git a/ui/app/_fallbacks/enterprise/lib/store/apis/scimApi.ts b/ui/app/_fallbacks/enterprise/lib/store/apis/scimApi.ts
index 1fc52da19a7..43cc4d8edc2 100644
--- a/ui/app/_fallbacks/enterprise/lib/store/apis/scimApi.ts
+++ b/ui/app/_fallbacks/enterprise/lib/store/apis/scimApi.ts
@@ -18,6 +18,9 @@ export const useGetAuthTypeQuery = (
// OSS stub for SCIM providers — returns an empty list so the onboarding
// widget's enterprise-only "configure SCIM" step is always considered
// incomplete (the step itself is hidden in OSS via IS_ENTERPRISE).
+// Element type mirrors the fields OSS consumers read off the enterprise
+// response (e.g. onboarding's `provider.enabled` gate); `unknown` elements
+// force every consumer into a per-site cast.
export const useGetSCIMProvidersQuery = (
_args?: undefined,
_opts?: { skip?: boolean },
diff --git a/ui/app/workspace/providers/fragments/apiKeysFormFragment.tsx b/ui/app/workspace/providers/fragments/apiKeysFormFragment.tsx
index 244c0fcb99d..41f0bf3e3f2 100644
--- a/ui/app/workspace/providers/fragments/apiKeysFormFragment.tsx
+++ b/ui/app/workspace/providers/fragments/apiKeysFormFragment.tsx
@@ -11,7 +11,7 @@ import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/comp
import { isRedacted } from "@/lib/utils/validation";
import { Info } from "lucide-react";
import { useEffect, useState } from "react";
-import { Control, UseFormReturn } from "react-hook-form";
+import { Control, UseFormReturn, useWatch } from "react-hook-form";
import { DeploymentsTable } from "./deploymentsTable";
// Providers that support batch APIs
@@ -23,6 +23,9 @@ interface Props {
// For custom providers, the underlying base provider type (e.g. "bedrock").
// Drives which credential UI renders; falls back to providerName for native providers.
baseProviderType?: string;
+ // The provider's configured network base URL (empty = provider default). Some
+ // per-provider toggles are only valid for specific upstream endpoint modes.
+ providerBaseURL?: string;
form: UseFormReturn;
}
@@ -135,7 +138,7 @@ function VPCEndpointsFormField({
);
}
-export function ApiKeyFormFragment({ control, providerName, baseProviderType, form }: Props) {
+export function ApiKeyFormFragment({ control, providerName, baseProviderType, providerBaseURL, form }: Props) {
// Credential UI keys off the base provider type for custom providers; the
// model list, deployments table, and API calls still use the real providerName.
const effectiveProvider = baseProviderType ?? providerName;
@@ -149,9 +152,21 @@ export function ApiKeyFormFragment({ control, providerName, baseProviderType, fo
const isSGL = effectiveProvider === "sgl";
const isDeepseek = effectiveProvider === "deepseek";
const isFireworks = effectiveProvider === "fireworks";
+ const isAlibaba = effectiveProvider === "alibaba";
+ const isKimi = effectiveProvider === "kimi";
+ const isZhipu = effectiveProvider === "zhipu";
const isKeylessProvider = isOllama || isSGL;
const supportsBatchAPI = BATCH_SUPPORTED_PROVIDERS.includes(effectiveProvider);
+ // Zhipu's Anthropic-compatible mount (…/api/anthropic) only authenticates GLM
+ // Coding Plan credentials. The provider default and every General API /paas/v4
+ // host derive the mount URL but get a 401 per request, so the Anthropic-endpoint
+ // toggles are gated to Coding Plan base URLs. Explicit exotic base URLs stay
+ // available (the backend derives /anthropic for them).
+ const zhipuBaseURL = (providerBaseURL ?? "").trim().replace(/\/+$/, "");
+ const isZhipuCodingPlanBaseURL = zhipuBaseURL.endsWith("/coding/paas/v4");
+ const isZhipuGeneralAPIBaseURL = isZhipu && !isZhipuCodingPlanBaseURL && (zhipuBaseURL === "" || zhipuBaseURL.endsWith("/paas/v4"));
+
// Auth type state for Azure: 'api_key', 'entra_id', or 'default_credential'
const [azureAuthType, setAzureAuthType] = useState<"api_key" | "entra_id" | "default_credential">("api_key");
@@ -204,6 +219,20 @@ export function ApiKeyFormFragment({ control, providerName, baseProviderType, fo
}
}, [isVertex, form]);
+ // A Zhipu General API base URL can never authenticate against the Anthropic
+ // mount, so a previously persisted toggle is cleared once the (possibly
+ // async) form.reset values arrive. Watching the field (instead of reading it
+ // in a one-shot mount effect) re-runs after reset populates edit values, and
+ // shouldDirty marks the correction saveable so Save is enabled.
+ const useAnthropicEndpoints = useWatch({ control, name: "key.use_anthropic_endpoints" });
+ useEffect(() => {
+ if (!isZhipuGeneralAPIBaseURL) return;
+ if (form.formState.isDirty) return;
+ if (useAnthropicEndpoints) {
+ form.setValue("key.use_anthropic_endpoints", false, { shouldDirty: true });
+ }
+ }, [isZhipuGeneralAPIBaseURL, useAnthropicEndpoints, form]);
+
useEffect(() => {
if (form.formState.isDirty) return;
if (isBedrock) {
@@ -457,6 +486,8 @@ export function ApiKeyFormFragment({ control, providerName, baseProviderType, fo
{
form.clearErrors("key.aliases");
@@ -855,7 +886,7 @@ export function ApiKeyFormFragment({ control, providerName, baseProviderType, fo
/>
)}
- {(isSGL || isDeepseek || isFireworks || isVLLM) && (
+ {(isSGL || isDeepseek || isFireworks || isVLLM || isAlibaba || isKimi || isZhipu) && (
Use Anthropic Endpoints
- Routes chat completions and responses requests through Anthropic-compatible endpoints.
+
+ {isZhipuGeneralAPIBaseURL
+ ? "Requires a GLM Coding Plan base URL (…/api/coding/paas/v4). The Anthropic-compatible endpoint rejects General API keys."
+ : "Routes chat completions and responses requests through Anthropic-compatible endpoints."}
+
diff --git a/ui/app/workspace/providers/fragments/deploymentsTable.tsx b/ui/app/workspace/providers/fragments/deploymentsTable.tsx
index 98b58db1b82..1134d47eb56 100644
--- a/ui/app/workspace/providers/fragments/deploymentsTable.tsx
+++ b/ui/app/workspace/providers/fragments/deploymentsTable.tsx
@@ -19,6 +19,14 @@ interface Props {
onChange: (next: Record) => void;
providerName: string;
disabled?: boolean;
+ // For custom providers, the underlying base provider type (e.g. "zhipu").
+ // Drives provider-specific deployment sections; providerName stays the real
+ // name for model lookup and display. Defaults to providerName.
+ baseProviderType?: string;
+ // True when the provider's base URL cannot authenticate against its
+ // Anthropic-compatible mount (currently Zhipu General API) — blocks the
+ // per-deployment "Use Anthropic endpoints" override from being set to On.
+ anthropicEndpointsDisabled?: boolean;
}
interface Row {
@@ -134,6 +142,7 @@ function TriStateOverrideRow({
value,
onChange,
disabled,
+ disableOn,
testId,
}: {
label: string;
@@ -141,6 +150,10 @@ function TriStateOverrideRow({
value: boolean | undefined;
onChange: (next: boolean | undefined) => void;
disabled?: boolean;
+ // Block the On choice (used when the provider's endpoint mode can't
+ // authenticate against the Anthropic mount) while still allowing remediation
+ // of a previously persisted On via Inherit/Off.
+ disableOn?: boolean;
testId?: string;
}) {
const id = useId();
@@ -162,7 +175,9 @@ function TriStateOverrideRow({
Use key setting
- On
+
+ On
+
Off
@@ -337,24 +352,44 @@ function ReplicateSection({ config, onChange, disabled }: ProviderSectionProps)
);
}
-function UseAnthropicEndpointsToggleSection({ config, onChange, disabled, providerName }: ProviderSectionProps & { providerName: string }) {
+function UseAnthropicEndpointsToggleSection({
+ config,
+ onChange,
+ disabled,
+ providerName,
+ anthropicEndpointsDisabled,
+}: ProviderSectionProps & { providerName: string; anthropicEndpointsDisabled?: boolean }) {
return (
onChange({ use_anthropic_endpoints: next })}
disabled={disabled}
+ disableOn={anthropicEndpointsDisabled}
testId="deployment-use-anthropic-endpoints"
/>
);
}
-function ProviderSection({ providerName, ...props }: ProviderSectionProps & { providerName: string }) {
- switch (providerName) {
+function ProviderSection({
+ providerName,
+ baseProviderType,
+ anthropicEndpointsDisabled,
+ ...props
+}: ProviderSectionProps & { providerName: string; baseProviderType?: string; anthropicEndpointsDisabled?: boolean }) {
+ // Provider-specific sections key off the base provider type so custom
+ // providers (e.g. "my-glm" on the zhipu base) get the right controls;
+ // providerName remains the real name for display and model lookup.
+ const effectiveProvider = baseProviderType ?? providerName;
+ switch (effectiveProvider) {
case "azure":
return ;
case "vertex":
@@ -373,6 +408,12 @@ function ProviderSection({ providerName, ...props }: ProviderSectionProps & { pr
return ;
case "vllm":
return ;
+ case "alibaba":
+ return ;
+ case "kimi":
+ return ;
+ case "zhipu":
+ return ;
default:
return null;
}
@@ -383,11 +424,15 @@ function ExpandedConfigPanel({
onChange,
providerName,
disabled,
+ baseProviderType,
+ anthropicEndpointsDisabled,
}: {
config: AliasConfig;
onChange: (patch: Partial) => void;
providerName: string;
disabled?: boolean;
+ baseProviderType?: string;
+ anthropicEndpointsDisabled?: boolean;
}) {
return (
@@ -432,12 +477,19 @@ function ExpandedConfigPanel({
/>
-
+
);
}
-export function DeploymentsTable({ value, onChange, providerName, disabled = false }: Props) {
+export function DeploymentsTable({ value, onChange, providerName, disabled = false, baseProviderType, anthropicEndpointsDisabled }: Props) {
const normalized = useMemo(() => normalize(value), [value]);
const rows: Row[] = useMemo(() => Object.entries(normalized).map(([name, config]) => ({ name, config })), [normalized]);
@@ -624,6 +676,8 @@ export function DeploymentsTable({ value, onChange, providerName, disabled = fal
onChange={(patch) => patchConfig(row.name, patch)}
providerName={providerName}
disabled={disabled}
+ baseProviderType={baseProviderType}
+ anthropicEndpointsDisabled={anthropicEndpointsDisabled}
/>
@@ -676,7 +730,14 @@ export function DeploymentsTable({ value, onChange, providerName, disabled = fal
)}
-
+
diff --git a/ui/app/workspace/providers/views/providerKeyForm.tsx b/ui/app/workspace/providers/views/providerKeyForm.tsx
index daf4b3d7547..a17a8f3fcfa 100644
--- a/ui/app/workspace/providers/views/providerKeyForm.tsx
+++ b/ui/app/workspace/providers/views/providerKeyForm.tsx
@@ -135,6 +135,7 @@ export default function ProviderKeyForm({ provider, keyId, onCancel, onSave }: P
control={form.control}
providerName={provider.name}
baseProviderType={provider.custom_provider_config?.base_provider_type}
+ providerBaseURL={provider.network_config?.base_url}
form={form}
/>
{isEditing && currentKey?.config_hash && }
diff --git a/ui/lib/constants/config.ts b/ui/lib/constants/config.ts
index 24a108506e3..a7757956e5c 100644
--- a/ui/lib/constants/config.ts
+++ b/ui/lib/constants/config.ts
@@ -66,6 +66,9 @@ export const ModelPlaceholders = {
fireworks: "e.g. accounts/fireworks/models/deepseek-v3p2",
sarvam: "e.g. sarvam-30b, sarvam-105b",
wafer: "e.g. glm-5.2, kimi-k2.6",
+ alibaba: "e.g. qwen3.8-max, qwen3.7-plus, qwen3.6-flash, text-embedding-v4",
+ kimi: "e.g. kimi-k3, kimi-k2.6, k3 (Kimi Code)",
+ zhipu: "e.g. glm-5.2, glm-4.7, glm-4.7-flash",
};
export const isKeyRequiredByProvider: Record = {
@@ -99,6 +102,9 @@ export const isKeyRequiredByProvider: Record = {
fireworks: true,
sarvam: true,
wafer: true,
+ alibaba: true,
+ kimi: true,
+ zhipu: true,
};
export const DefaultNetworkConfig = {
diff --git a/ui/lib/constants/icons.tsx b/ui/lib/constants/icons.tsx
index 1caf4a352e1..e0caf5554f4 100644
--- a/ui/lib/constants/icons.tsx
+++ b/ui/lib/constants/icons.tsx
@@ -818,6 +818,57 @@ export const ProviderIcons = {
);
},
+ alibaba: ({ size = "md", className = "" }: IconProps) => {
+ const resolvedSize = resolveSize(size);
+
+ return (
+
+ Alibaba Cloud
+
+
+ );
+ },
+ kimi: ({ size = "md", className = "" }: IconProps) => {
+ const resolvedSize = resolveSize(size);
+
+ return (
+
+ Kimi
+
+
+ );
+ },
+ zhipu: ({ size = "md", className = "" }: IconProps) => {
+ const resolvedSize = resolveSize(size);
+
+ return (
+
+ Zhipu AI
+
+
+ );
+ },
} as const;
// Routing Engine Icons
diff --git a/ui/lib/constants/logs.ts b/ui/lib/constants/logs.ts
index 2ea9a164836..f731285222f 100644
--- a/ui/lib/constants/logs.ts
+++ b/ui/lib/constants/logs.ts
@@ -30,6 +30,9 @@ export const KnownProvidersNames = [
"fireworks",
"sarvam",
"wafer",
+ "alibaba",
+ "kimi",
+ "zhipu",
] as const;
// Local Provider type derived from KNOWN_PROVIDERS constant
@@ -40,6 +43,7 @@ export const ProviderNames: readonly ProviderName[] = KnownProvidersNames;
// Built-in providers whose Bifrost implementation supports embedding requests.
// Custom providers must instead be checked via custom_provider_config.allowed_requests.embedding.
export const EmbeddingSupportedProviders: readonly ProviderName[] = [
+ "alibaba",
"azure",
"bedrock",
"cohere",
@@ -156,6 +160,9 @@ export const ProviderLabels: Record = {
fireworks: "Fireworks AI",
sarvam: "Sarvam AI",
wafer: "Wafer",
+ alibaba: "Alibaba Cloud",
+ kimi: "Kimi",
+ zhipu: "Zhipu AI",
} as const;
// Helper function to get provider label, supporting custom providers