Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 12 additions & 4 deletions core/providers/bedrock/bedrock.go
Original file line number Diff line number Diff line change
Expand Up @@ -1276,9 +1276,13 @@ func (provider *BedrockProvider) ChatCompletion(ctx *schemas.BifrostContext, key
return nil, err
}

if provider.routesToMantle(ctx, key, request.Model) {
surface := provider.resolveSurface(ctx, key, request.Model)
if surface.isMantle() {
return provider.mantleChatCompletions(ctx, key, request)
}
if runtimeServesOpenAIAPI(ctx, key, surface, request.Model, schemas.BedrockAPIChatCompletions) {
return provider.runtimeChatCompletions(ctx, key, request)
}

// Use Bedrock Converse API for all other models
jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody(
Expand Down Expand Up @@ -1467,9 +1471,13 @@ func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContex
return nil, err
}

if provider.routesToMantle(ctx, key, request.Model) {
surface := provider.resolveSurface(ctx, key, request.Model)
if surface.isMantle() {
return provider.mantleChatCompletionsStream(ctx, postHookRunner, postHookSpanFinalizer, key, request)
}
if runtimeServesOpenAIAPI(ctx, key, surface, request.Model, schemas.BedrockAPIChatCompletions) {
return provider.runtimeChatCompletionsStream(ctx, postHookRunner, postHookSpanFinalizer, key, request)
}

// Use Bedrock Converse streaming API for all other models
jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody(
Expand Down Expand Up @@ -1793,7 +1801,7 @@ func (provider *BedrockProvider) Responses(ctx *schemas.BifrostContext, key sche
if surface.isMantle() {
return provider.mantleResponses(ctx, key, request)
}
if runtimeServesResponses(ctx, surface, request.Model) {
if runtimeServesOpenAIAPI(ctx, key, surface, request.Model, schemas.BedrockAPIResponses) {
return provider.runtimeResponses(ctx, key, request)
}

Expand Down Expand Up @@ -1883,7 +1891,7 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po
if surface.isMantle() {
return provider.mantleResponsesStream(ctx, postHookRunner, postHookSpanFinalizer, key, request)
}
if runtimeServesResponses(ctx, surface, request.Model) {
if runtimeServesOpenAIAPI(ctx, key, surface, request.Model, schemas.BedrockAPIResponses) {
return provider.runtimeResponsesStream(ctx, postHookRunner, postHookSpanFinalizer, key, request)
}

Expand Down
71 changes: 71 additions & 0 deletions core/providers/bedrock/runtimeopenai.go
Original file line number Diff line number Diff line change
Expand Up @@ -90,3 +90,74 @@ func (provider *BedrockProvider) runtimeResponsesStream(
postHookSpanFinalizer,
)
}

// runtimeChatCompletions handles non-streaming chat requests on bedrock-runtime's
// OpenAI-compatible surface. Reached only when the operator opts in.
func (provider *BedrockProvider) runtimeChatCompletions(
ctx *schemas.BifrostContext,
key schemas.Key,
request *schemas.BifrostChatRequest,
) (*schemas.BifrostChatResponse, *schemas.BifrostError) {
region := resolveBedrockRegion(ctx, key, request.Model)
url := runtimeOpenAIURL(bedrockEndpoints(key.BedrockKeyConfig), region, "chat/completions")

var signer providerUtils.BodySigner
if key.Value.GetValue() == "" {
signer = func(body []byte) (map[string]string, *schemas.BifrostError) {
return signOpenAIV4Headers(ctx, body, url, "application/json", key, region, provider.networkConfig.ExtraHeaders, bedrockSigningService)
}
}

return openai.HandleOpenAIChatCompletionRequest(
ctx,
provider.mantleClient,
url,
request,
openai.BearerAuthHeader(key),
provider.networkConfig.ExtraHeaders,
providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
provider.GetProviderKey(),
nil,
nil,
signer,
provider.logger,
)
}

// runtimeChatCompletionsStream handles streaming chat requests on bedrock-runtime's
// OpenAI-compatible surface.
func (provider *BedrockProvider) runtimeChatCompletionsStream(
ctx *schemas.BifrostContext,
postHookRunner schemas.PostHookRunner,
postHookSpanFinalizer func(context.Context),
key schemas.Key,
request *schemas.BifrostChatRequest,
) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) {
region := resolveBedrockRegion(ctx, key, request.Model)
url := runtimeOpenAIURL(bedrockEndpoints(key.BedrockKeyConfig), region, "chat/completions")

var signer providerUtils.BodySigner
if key.Value.GetValue() == "" {
signer = func(body []byte) (map[string]string, *schemas.BifrostError) {
return signOpenAIV4Headers(ctx, body, url, "text/event-stream", key, region, provider.networkConfig.ExtraHeaders, bedrockSigningService)
}
}

return openai.HandleOpenAIChatCompletionStreaming(
ctx, provider.mantleStreamingClient, url, request,
openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders,
provider.networkConfig.StreamIdleTimeoutInSeconds,
providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest),
providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse),
provider.GetProviderKey(), postHookRunner,
nil,
nil,
nil,
nil,
nil,
signer,
provider.logger,
postHookSpanFinalizer,
)
}
40 changes: 30 additions & 10 deletions core/providers/bedrock/surface.go
Original file line number Diff line number Diff line change
Expand Up @@ -169,24 +169,44 @@ func resolveBedrockSurface(ctx *schemas.BifrostContext, key schemas.Key, model s
return bedrockSurface{host: bedrockServiceRuntime, reason: reasonModelFamilyFallback}
}

// runtimeServesResponses reports whether a runtime-bound Responses request should use
// bedrock-runtime's OpenAI-compatible /openai/v1/responses surface instead of Converse.
// ResolveUseOpenAIEndpoints reports whether this key or alias routes Bedrock inference
// through the OpenAI-compatible endpoints instead of Converse. Opt-in, and an alias value
// wins over the key, mirroring ResolveUseAnthropicEndpoints.
//
// Converse holds no conversation state and has no previous_response_id, so it silently
// drops the reference: a stateful client sends only the new turn and the model never sees
// the rest. With tools that fails outright, since the tool result arrives with its toolUse
// left behind in state. The OpenAI surface both reads and mints response ids.
// Opt-in rather than automatic because the two surfaces are not interchangeable. Converse
// carries Bedrock Guardrails, performanceConfig and requestMetadata, all of which the
// OpenAI-compatible endpoints accept and silently ignore, so diverting on Bifrost's own
// initiative could stop a guardrail being enforced with no error anywhere.
func ResolveUseOpenAIEndpoints(ctx *schemas.BifrostContext, key schemas.Key) bool {
if ra := schemas.GetResolvedAlias(ctx); ra != nil && ra.Config != nil && ra.Config.UseOpenAIEndpoints != nil {
return *ra.Config.UseOpenAIEndpoints
}
return key.UseOpenAIEndpoints != nil && *key.UseOpenAIEndpoints
}

// runtimeServesOpenAIAPI reports whether a runtime-bound request should use
// bedrock-runtime's OpenAI-compatible surface for the given wire API.
//
// The datasheet decides when it publishes a runtime row; otherwise family detection, since
// AWS 404s every other family on this path ("doesn't support this API").
func runtimeServesResponses(ctx *schemas.BifrostContext, surface bedrockSurface, model string) bool {
// What the opt-in buys: Converse holds no conversation state and has no
// previous_response_id, so it silently drops the reference. A stateful client sends only
// the new turn and the model never sees the rest; with tools that fails outright, since
// the tool result arrives with its toolUse left behind in state.
//
// The datasheet decides support when it publishes a runtime row; otherwise family
// detection, since AWS 404s every other family here ("doesn't support this API"). Support
// is a separate question from the flag: opting in never forces a surface the model cannot
// serve.
func runtimeServesOpenAIAPI(ctx *schemas.BifrostContext, key schemas.Key, surface bedrockSurface, model string, api schemas.BedrockAPI) bool {
if !ResolveUseOpenAIEndpoints(ctx, key) {
return false
}
// An application inference profile is Converse-only, so it must never divert.
if surface.isMantle() || surface.reason == reasonApplicationProfile {
return false
}
canonical := schemas.ResolveCanonicalModel(ctx, model)
if apis := schemas.ResolveModelCaps(schemas.Bedrock, canonical).BedrockAPIs(); len(apis) > 0 {
return slices.Contains(apis, schemas.BedrockAPIResponses)
return slices.Contains(apis, api)
}
return schemas.IsOpenAIModelFamily(ctx, canonical) || schemas.IsGrokModel(canonical)
}
Expand Down
95 changes: 83 additions & 12 deletions core/providers/bedrock/surface_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -599,6 +599,7 @@ func TestIsAIPResourceID(t *testing.T) {
// detection is what actually gates the OpenAI-compatible Responses surface
// today. AWS 404s every other family there.
func TestRuntimeServesResponsesFamilyFallback(t *testing.T) {
optedIn := schemas.Key{UseOpenAIEndpoints: schemas.Ptr(true)}
cases := []struct {
model string
want bool
Expand All @@ -613,9 +614,9 @@ func TestRuntimeServesResponsesFamilyFallback(t *testing.T) {
for _, tc := range cases {
t.Run(tc.model, func(t *testing.T) {
ctx := surfaceTestCtx()
surface := resolveBedrockSurface(ctx, schemas.Key{}, tc.model)
if got := runtimeServesResponses(ctx, surface, tc.model); got != tc.want {
t.Errorf("runtimeServesResponses(%q) = %v, want %v", tc.model, got, tc.want)
surface := resolveBedrockSurface(ctx, optedIn, tc.model)
if got := runtimeServesOpenAIAPI(ctx, optedIn, surface, tc.model, schemas.BedrockAPIResponses); got != tc.want {
t.Errorf("runtimeServesOpenAIAPI(%q) = %v, want %v", tc.model, got, tc.want)
}
})
}
Expand All @@ -624,6 +625,7 @@ func TestRuntimeServesResponsesFamilyFallback(t *testing.T) {
// A published runtime row is authoritative in both directions: it can divert a
// model family detection would not, and hold back one it would.
func TestRuntimeServesResponsesDatasheetWinsOverFamily(t *testing.T) {
optedIn := schemas.Key{UseOpenAIEndpoints: schemas.Ptr(true)}
installCaps(t, map[schemas.ModelProvider]map[string][]schemas.BedrockAPI{
schemas.Bedrock: {
// Converse-only despite being OpenAI family.
Expand All @@ -642,9 +644,9 @@ func TestRuntimeServesResponsesDatasheetWinsOverFamily(t *testing.T) {
for _, tc := range cases {
t.Run(tc.model, func(t *testing.T) {
ctx := surfaceTestCtx()
surface := resolveBedrockSurface(ctx, schemas.Key{}, tc.model)
if got := runtimeServesResponses(ctx, surface, tc.model); got != tc.want {
t.Errorf("runtimeServesResponses(%q) = %v, want %v", tc.model, got, tc.want)
surface := resolveBedrockSurface(ctx, optedIn, tc.model)
if got := runtimeServesOpenAIAPI(ctx, optedIn, surface, tc.model, schemas.BedrockAPIResponses); got != tc.want {
t.Errorf("runtimeServesOpenAIAPI(%q) = %v, want %v", tc.model, got, tc.want)
}
})
}
Expand All @@ -654,40 +656,44 @@ func TestRuntimeServesResponsesDatasheetWinsOverFamily(t *testing.T) {
// family, so without the guard the alias chain would resolve it to an OpenAI
// name and divert a request AWS cannot serve.
func TestRuntimeServesResponsesNeverDivertsApplicationProfile(t *testing.T) {
optedIn := keyWithARN(appProfileARN)
optedIn.UseOpenAIEndpoints = schemas.Ptr(true)
ctx := withAlias("my-gpt", "3dnkdwuaalc7", appProfileARN)
name := "gpt-5.6-luna"
schemas.GetResolvedAlias(ctx).Config.ModelName = &name

surface := resolveBedrockSurface(ctx, keyWithARN(appProfileARN), "3dnkdwuaalc7")
surface := resolveBedrockSurface(ctx, optedIn, "3dnkdwuaalc7")
if surface.reason != reasonApplicationProfile {
t.Fatalf("precondition: reason = %q, want %q", surface.reason, reasonApplicationProfile)
}
if runtimeServesResponses(ctx, surface, "3dnkdwuaalc7") {
if runtimeServesOpenAIAPI(ctx, optedIn, surface, "3dnkdwuaalc7", schemas.BedrockAPIResponses) {
t.Error("an application inference profile must stay on Converse")
}
}

// Mantle has its own Responses path; the runtime surface must never claim it.
func TestRuntimeServesResponsesIgnoresMantleSurface(t *testing.T) {
optedIn := schemas.Key{UseOpenAIEndpoints: schemas.Ptr(true)}
ctx := surfaceTestCtx()
surface := resolveBedrockSurface(ctx, schemas.Key{}, "openai.gpt-5.6-terra")
surface := resolveBedrockSurface(ctx, optedIn, "openai.gpt-5.6-terra")
if !surface.isMantle() {
t.Fatalf("precondition: bare id should route to mantle, got %q", surface.host)
}
if runtimeServesResponses(ctx, surface, "openai.gpt-5.6-terra") {
if runtimeServesOpenAIAPI(ctx, optedIn, surface, "openai.gpt-5.6-terra", schemas.BedrockAPIResponses) {
t.Error("a mantle-bound request must not divert to the runtime surface")
}
}

// The gate reads the canonical name, so an alias whose wire id carries no family
// still diverts.
func TestRuntimeServesResponsesResolvesAliasedModel(t *testing.T) {
optedIn := schemas.Key{UseOpenAIEndpoints: schemas.Ptr(true)}
ctx := withAlias("my-gpt", "us.openai.gpt-5.6-terra", "")
name := "gpt-5.6-terra"
schemas.GetResolvedAlias(ctx).Config.ModelName = &name

surface := resolveBedrockSurface(ctx, schemas.Key{}, "us.openai.gpt-5.6-terra")
if !runtimeServesResponses(ctx, surface, "us.openai.gpt-5.6-terra") {
surface := resolveBedrockSurface(ctx, optedIn, "us.openai.gpt-5.6-terra")
if !runtimeServesOpenAIAPI(ctx, optedIn, surface, "us.openai.gpt-5.6-terra", schemas.BedrockAPIResponses) {
t.Error("aliased OpenAI model must divert to the runtime Responses surface")
}
}
Expand All @@ -697,3 +703,68 @@ func TestRuntimeOpenAIURL(t *testing.T) {
t.Errorf("runtimeOpenAIURL = %q", got)
}
}

// Opt-in, two states, mirroring use_anthropic_endpoints: off keeps everything on
// Converse, on moves both request types to the OpenAI-compatible surface.
func TestUseOpenAIEndpointsFlag(t *testing.T) {
const model = "us.openai.gpt-5.6-terra"
cases := []struct {
name string
flag *bool
want bool
}{
{"unset", nil, false},
{"false", schemas.Ptr(false), false},
{"true", schemas.Ptr(true), true},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
ctx := surfaceTestCtx()
key := schemas.Key{UseOpenAIEndpoints: tc.flag}
surface := resolveBedrockSurface(ctx, key, model)
for _, api := range []schemas.BedrockAPI{schemas.BedrockAPIResponses, schemas.BedrockAPIChatCompletions} {
if got := runtimeServesOpenAIAPI(ctx, key, surface, model, api); got != tc.want {
t.Errorf("%s = %v, want %v", api, got, tc.want)
}
}
})
}
}

// An alias-level value wins over the key, matching use_anthropic_endpoints.
func TestUseOpenAIEndpointsAliasOverridesKey(t *testing.T) {
const model = "us.openai.gpt-5.6-terra"
ctx := withAlias("my-gpt", model, "")
schemas.GetResolvedAlias(ctx).Config.UseOpenAIEndpoints = schemas.Ptr(false)

key := schemas.Key{UseOpenAIEndpoints: schemas.Ptr(true)}
surface := resolveBedrockSurface(ctx, key, model)
if runtimeServesOpenAIAPI(ctx, key, surface, model, schemas.BedrockAPIResponses) {
t.Error("alias false must override key true")
}
}

// Opting in cannot force a surface the model does not serve: AWS 404s Claude there.
func TestUseOpenAIEndpointsCannotForceUnsupportedModel(t *testing.T) {
const model = "us.anthropic.claude-sonnet-4-6"
key := schemas.Key{UseOpenAIEndpoints: schemas.Ptr(true)}
ctx := surfaceTestCtx()
surface := resolveBedrockSurface(ctx, key, model)
if runtimeServesOpenAIAPI(ctx, key, surface, model, schemas.BedrockAPIResponses) {
t.Error("claude must stay on converse even when opted in")
}
}

// Nor can it override the application-inference-profile guard.
func TestUseOpenAIEndpointsCannotForceApplicationProfile(t *testing.T) {
ctx := withAlias("my-gpt", "3dnkdwuaalc7", appProfileARN)
name := "gpt-5.6-luna"
schemas.GetResolvedAlias(ctx).Config.ModelName = &name

key := keyWithARN(appProfileARN)
key.UseOpenAIEndpoints = schemas.Ptr(true)
surface := resolveBedrockSurface(ctx, key, "3dnkdwuaalc7")
if runtimeServesOpenAIAPI(ctx, key, surface, "3dnkdwuaalc7", schemas.BedrockAPIResponses) {
t.Error("an application inference profile must stay on Converse even when opted in")
}
}
8 changes: 5 additions & 3 deletions core/providers/openai/responses.go
Original file line number Diff line number Diff line change
Expand Up @@ -153,15 +153,17 @@ const maxResponsesCacheBreakpoints = 4
// per-block cache_control through /v1/responses and converts a breakpoint back into
// an Anthropic one (#6290). OpenAI defined the field for gpt-5.6, where it pairs with
// request-level prompt_cache_options; Azure and Bedrock Mantle serve the same models
// through the same wire format, so they inherit it (#6180).
// through the same wire format, so they inherit it (#6180). Bedrock is listed for the
// same reason: its OpenAI-compatible surfaces on both hosts speak that wire format, and
// a request there reports the bedrock key rather than bedrock_mantle.
//
// Everything else either accepts cache_control directly or caches implicitly, and for
// those the serializer's existing strip is the correct behaviour.
func responsesUsesPromptCacheBreakpoints(provider schemas.ModelProvider, model string) bool {
switch provider {
case schemas.OpenRouter:
return true
case schemas.OpenAI, schemas.Azure, schemas.BedrockMantle:
case schemas.OpenAI, schemas.Azure, schemas.BedrockMantle, schemas.Bedrock:
return schemas.IsGPT56Model(model)
default:
return false
Expand Down Expand Up @@ -195,7 +197,7 @@ func responsesHasPromptCacheBreakpoint(messages []schemas.ResponsesMessage) bool
// that off; mode=explicit does. OpenRouter has no equivalent field and needs none.
func responsesUsesPromptCacheOptions(provider schemas.ModelProvider, model string) bool {
switch provider {
case schemas.OpenAI, schemas.Azure, schemas.BedrockMantle:
case schemas.OpenAI, schemas.Azure, schemas.BedrockMantle, schemas.Bedrock:
return schemas.IsGPT56Model(model)
default:
return false
Expand Down
3 changes: 3 additions & 0 deletions core/schemas/account.go
Original file line number Diff line number Diff line change
Expand Up @@ -144,6 +144,7 @@ type Key struct {
Enabled *bool `json:"enabled,omitempty"` // Whether the key is active (default:true)
UseForBatchAPI *bool `json:"use_for_batch_api,omitempty"` // Whether this key can be used for batch API operations (default:false for new keys, migrated keys default to true)
UseAnthropicEndpoints *bool `json:"use_anthropic_endpoints,omitempty"` // Whether to use anthropic endpoints for this key
UseOpenAIEndpoints *bool `json:"use_openai_endpoints,omitempty"` // Whether to use OpenAI-compatible endpoints for this key
ConfigHash string `json:"config_hash,omitempty"` // Hash of config.json version, used for change detection
Status KeyStatusType `json:"status,omitempty"` // Status of key
Description string `json:"description,omitempty"` // Description of key
Expand Down Expand Up @@ -234,6 +235,7 @@ type AliasConfig struct {
// a field name shared by multiple same-depth anonymous structs.
ProjectID *SecretVar `json:"project_id,omitempty"`
UseAnthropicEndpoints *bool `json:"use_anthropic_endpoints,omitempty"` // Whether to use anthropic endpoints for this alias
UseOpenAIEndpoints *bool `json:"use_openai_endpoints,omitempty"` // Whether to use OpenAI-compatible endpoints for this alias

*AzureAliasCfg
*VertexAliasCfg
Expand All @@ -252,6 +254,7 @@ func (ac AliasConfig) isLegacyShape() bool {
ac.Region == nil &&
ac.ProjectID == nil &&
ac.UseAnthropicEndpoints == nil &&
ac.UseOpenAIEndpoints == nil &&
ac.AzureAliasCfg == nil &&
ac.VertexAliasCfg == nil &&
ac.BedrockAliasCfg == nil &&
Expand Down
Loading
Loading